@cspeach/cli 1.0.0 → 1.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/loop.js +22 -9
- package/dist/approvals/op-labels.js +124 -0
- package/dist/approvals/render.js +42 -36
- package/dist/cli.js +15 -0
- package/dist/commands/compact.js +28 -2
- package/dist/commands/config-set.js +189 -0
- package/dist/commands/config-show.js +20 -0
- package/dist/commands/export-audit.js +43 -0
- package/dist/commands/help.js +5 -0
- package/dist/commands/plan-audit-evidence.js +266 -0
- package/dist/commands/plan-audit.js +692 -0
- package/dist/commands/plan-chain.js +671 -0
- package/dist/commands/plan-continue.js +179 -0
- package/dist/commands/plan-gate.js +154 -0
- package/dist/commands/plan-resume.js +588 -33
- package/dist/config/loader.js +128 -4
- package/dist/config/model-defaults.js +14 -0
- package/dist/cost/pricing.js +27 -1
- package/dist/doctor/checks/system-roles.js +41 -0
- package/dist/doctor/run.js +2 -0
- package/dist/models/resolve.js +61 -0
- package/dist/models/server-config.js +155 -0
- package/dist/one-shot.js +25 -3
- package/dist/projects/extract-cca.js +3 -1
- package/dist/projects/extract-modernize.js +3 -1
- package/dist/projects/extract-plan.js +60 -6
- package/dist/projects/extract-test-coverage.js +3 -1
- package/dist/projects/extract-upgrade.js +3 -1
- package/dist/projects/handover-md.js +195 -0
- package/dist/projects/index.js +1 -1
- package/dist/projects/plan-run.js +137 -13
- package/dist/projects/plan-schema.js +73 -0
- package/dist/projects/run-lease.js +157 -0
- package/dist/projects/save-command.js +26 -15
- package/dist/renderer/status-footer.js +22 -12
- package/dist/renderer/thinking-heartbeat.js +64 -8
- package/dist/renderer/todo-block.js +51 -0
- package/dist/renderer/tool-widget.js +37 -0
- package/dist/repl/bracketed-paste.js +28 -19
- package/dist/repl/builtin-commands.js +5 -0
- package/dist/repl/current-transport.js +10 -0
- package/dist/repl/history.js +86 -0
- package/dist/repl/ink-stdin-guard.js +64 -0
- package/dist/repl/mode-ceiling.js +16 -0
- package/dist/repl/mode-cycle.js +104 -0
- package/dist/repl/post-turn-status.js +24 -4
- package/dist/repl/slash-completer.js +5 -0
- package/dist/repl.js +954 -83
- package/dist/rewind/candidates.js +194 -0
- package/dist/rewind/cli.js +137 -0
- package/dist/rewind/format.js +27 -0
- package/dist/rewind/restore.js +245 -0
- package/dist/session/audit-export.js +459 -0
- package/dist/session/context-report.js +163 -0
- package/dist/session/recap.js +160 -0
- package/dist/skill-catalog.js +9 -3
- package/dist/skills/bundled-skills.js +71 -78
- package/dist/tools/approval.js +115 -7
- package/dist/tools/ask-question.js +304 -3
- package/dist/tools/extend-model/anchored-insert.js +604 -0
- package/dist/tools/extend-model/tool.js +162 -10
- package/dist/tools/fiori/fe-extend.js +76 -0
- package/dist/tools/fiori/fe-scaffold.js +29 -3
- package/dist/tools/fiori/floorplan-map.js +19 -0
- package/dist/tools/fiori/samples/data/index.json +13602 -0
- package/dist/tools/fiori/samples/data/sources.generated.js +808 -0
- package/dist/tools/fiori/samples/loader.js +248 -0
- package/dist/tools/fiori/samples/search.js +63 -0
- package/dist/tools/fiori/samples/types.js +2 -0
- package/dist/tools/fiori/smoke/assertions.js +74 -0
- package/dist/tools/fiori/smoke/browser.js +52 -0
- package/dist/tools/fiori/smoke/driver.js +89 -0
- package/dist/tools/fiori/smoke/freestyle-spec.js +317 -0
- package/dist/tools/fiori/smoke/run-smoke.js +149 -0
- package/dist/tools/fiori/tools.js +328 -3
- package/dist/tools/local-build.js +11 -1
- package/dist/tools/sap-read.js +79 -11
- package/dist/tools/sap-write.js +24 -4
- package/dist/tools/snapshot.js +27 -1
- package/dist/tools/subagent/agent_run.js +27 -3
- package/dist/tools/todo.js +144 -0
- package/dist/ui/app.js +372 -19
- package/dist/ui/approval-modal.js +49 -16
- package/dist/ui/ask-question-emitter.js +14 -0
- package/dist/ui/context-grid.js +108 -0
- package/dist/ui/footer.js +109 -30
- package/dist/ui/header.js +7 -0
- package/dist/ui/line-resolution.js +18 -2
- package/dist/ui/rewind-emitter.js +10 -0
- package/dist/ui/rewind-panel.js +81 -0
- package/dist/ui/sap-state-store.js +1 -0
- package/dist/ui/status-line.js +43 -0
- package/dist/ui/text-input.js +72 -8
- package/dist/ui/todo-emitter.js +25 -0
- package/dist/ui/todo-panel.js +64 -0
- package/dist/ui/turn-status-emitter.js +50 -4
- package/dist/ui/turn-status.js +18 -3
- package/dist/ui/widgets/ask-form.js +242 -0
- package/dist/ui/widgets/ask-question-modal.js +17 -7
- package/package.json +4 -1
|
@@ -0,0 +1,692 @@
|
|
|
1
|
+
// cspeach-cli/src/commands/plan-audit.ts
|
|
2
|
+
//
|
|
3
|
+
// Task 5 (agentic-flow, 2026-07-03) — the auditor turn.
|
|
4
|
+
//
|
|
5
|
+
// After a plan phase executes and its self-reported manifest is persisted,
|
|
6
|
+
// this module runs a FRESH-CONTEXT, cheap-tier, tool-less model turn that
|
|
7
|
+
// judges the phase's self-report against ground truth the audited model did
|
|
8
|
+
// NOT author: the bounded write/verify tool-call excerpt from
|
|
9
|
+
// plan-audit-evidence.ts. The phase model grading its own homework is the
|
|
10
|
+
// exact failure mode this exists to close.
|
|
11
|
+
//
|
|
12
|
+
// Isolation properties (same mechanism as tools/subagent/agent_run.ts):
|
|
13
|
+
// - fresh SessionState per attempt (no shared transcript, no id collision)
|
|
14
|
+
// - null-sink EventEmitter (audit output never reaches the user's UI;
|
|
15
|
+
// Task 6 renders the parsed verdict, not the raw turn)
|
|
16
|
+
// - toolFilter: () => false — a fully tool-less turn (stricter than
|
|
17
|
+
// agent_run's read-only case; the tools array sent to the LLM is empty),
|
|
18
|
+
// PLUS a prompt instruction not to call tools (belt and braces)
|
|
19
|
+
// - modelOverride: the shared Sonnet-tier constant from plan-model-tier.ts
|
|
20
|
+
// (never hardcoded here; Haiku stays unreachable by construction)
|
|
21
|
+
// - suppressSaveHook: the audit turn must never offer to save an envelope
|
|
22
|
+
//
|
|
23
|
+
// Failure containment: runPhaseAudit NEVER throws. Thrown errors and
|
|
24
|
+
// unparseable output are retried (3 attempts total); exhaustion returns
|
|
25
|
+
// { state: 'infra_failed', ... } which Task 6's gate treats as a stop —
|
|
26
|
+
// an audit that could not run is not a pass.
|
|
27
|
+
//
|
|
28
|
+
// This module must NOT import repl.tsx or any Ink/UI module — Task 6 owns
|
|
29
|
+
// the wiring and rendering.
|
|
30
|
+
import { EventEmitter } from 'node:events';
|
|
31
|
+
import { randomBytes } from 'node:crypto';
|
|
32
|
+
import { runTurn } from '../agent/loop.js';
|
|
33
|
+
import { collectTurnAssistantText } from '../agent/turn-assistant-text.js';
|
|
34
|
+
import { createProviderForMode } from '../agent/providers/factory.js';
|
|
35
|
+
import { loadConfig } from '../config/loader.js';
|
|
36
|
+
import { getStore } from '../auth/api-key.js';
|
|
37
|
+
import { newSession } from '../session/schema.js';
|
|
38
|
+
import { parsePlanContent } from '../projects/plan-schema.js';
|
|
39
|
+
import { statusesFromItems } from '../projects/plan-run.js';
|
|
40
|
+
import { extractAuditEvidence } from './plan-audit-evidence.js';
|
|
41
|
+
import { resolveModelRole } from '../models/resolve.js';
|
|
42
|
+
/** 3 attempts total = 1 try + up to 2 retries. */
|
|
43
|
+
const MAX_ATTEMPTS = 3;
|
|
44
|
+
/**
|
|
45
|
+
* Resolve the model the auditor turn runs on. Precedence:
|
|
46
|
+
* 1. env CSPEACH_AUDIT_MODEL (ops escape hatch — no format check, trimmed)
|
|
47
|
+
* 2. local config `audit_model` (persistent opt-in, trimmed)
|
|
48
|
+
* 3. PLAN_TIER_SONNET_MODEL (today's hardcoded default — nothing set ⇒
|
|
49
|
+
* byte-identical to before this feature)
|
|
50
|
+
*
|
|
51
|
+
* The B1 fix (docs/design/2026-07-09-session-findings-and-backlog.md §D, Step 1):
|
|
52
|
+
* the audit intermittently STALLED on the Sonnet route; pointing it at a
|
|
53
|
+
* reliable model stops the stall. Pure — no I/O; the caller passes a loaded
|
|
54
|
+
* config. Empty/whitespace at either source is treated as unset so a blank env
|
|
55
|
+
* or a `audit_model = ""` typo can never resolve to an invalid model id.
|
|
56
|
+
*/
|
|
57
|
+
export function resolveAuditModel(cfg, env = process.env) {
|
|
58
|
+
// model-governance step 2d — the exported name + signature stay (three
|
|
59
|
+
// consumers + existing tests depend on it), but the logic delegates to the
|
|
60
|
+
// generalized resolver so a served roles.audit slots into precedence BELOW
|
|
61
|
+
// local config and ABOVE the Sonnet built-in. Only `audit_model` is read for
|
|
62
|
+
// the audit role, so the partial cfg is sufficient.
|
|
63
|
+
return resolveModelRole('audit', cfg, env);
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* audit-timeout (2026-07-06) — hard per-attempt deadline.
|
|
67
|
+
*
|
|
68
|
+
* Live defect: the audit turn hung FOREVER — the proxy model call under
|
|
69
|
+
* runPhaseAudit never returned and nothing aborted it (observed stuck at 5m+
|
|
70
|
+
* with the heartbeat still ticking; the owner had to Ctrl+C). runTurn accepts
|
|
71
|
+
* an AbortSignal, but nobody was firing it. This deadline guarantees every
|
|
72
|
+
* attempt terminates: a timed-out attempt counts as a failed attempt → retry
|
|
73
|
+
* in a fresh session (as already designed) → after 3, the existing infra_failed
|
|
74
|
+
* path (banner, resume re-audits). Never a hang.
|
|
75
|
+
*
|
|
76
|
+
* 120s is deliberately generous: the successful live audits ran 5-40s, and a
|
|
77
|
+
* big-evidence audit (a 23k-token run was observed) can legitimately need more,
|
|
78
|
+
* so 120s is 3-24× headroom over the observed successes while still bounding the
|
|
79
|
+
* whole audit at 120s×3 = 360s (~6m) worst case. Overridable via
|
|
80
|
+
* CSPEACH_AUDIT_ATTEMPT_TIMEOUT_MS (ops escape hatch) or the per-call
|
|
81
|
+
* attemptTimeoutMs arg (tests inject a tiny value).
|
|
82
|
+
*/
|
|
83
|
+
export const AUDIT_ATTEMPT_TIMEOUT_MS = 120_000;
|
|
84
|
+
function resolveAttemptTimeoutMs(override) {
|
|
85
|
+
if (typeof override === 'number' && Number.isFinite(override) && override > 0)
|
|
86
|
+
return override;
|
|
87
|
+
const raw = process.env.CSPEACH_AUDIT_ATTEMPT_TIMEOUT_MS;
|
|
88
|
+
if (raw) {
|
|
89
|
+
const n = Number.parseInt(raw, 10);
|
|
90
|
+
if (Number.isFinite(n) && n > 0)
|
|
91
|
+
return n;
|
|
92
|
+
}
|
|
93
|
+
return AUDIT_ATTEMPT_TIMEOUT_MS;
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* Skill header for the audit turn. The audit is part of the plan flow and
|
|
97
|
+
* the proxy requires a valid catalog skill; the prompt contract below is
|
|
98
|
+
* deliberately strong enough to override playbook-shaped instructions.
|
|
99
|
+
*/
|
|
100
|
+
const AUDIT_SKILL = 'abap-plan';
|
|
101
|
+
/**
|
|
102
|
+
* The auditor's instructions — exported as a named constant so tests pin the
|
|
103
|
+
* contract instead of regexing prose. Appended verbatim to every audit prompt.
|
|
104
|
+
*
|
|
105
|
+
* Confidence-tiered redesign (2026-07-08, docs/design/2026-07-08-audit-
|
|
106
|
+
* confidence-tiered-redesign.md §2–§6). The auditor is no longer a binary
|
|
107
|
+
* pass/fail judge that DEFAULTS TO FAIL: it BLOCKS (verdict "fail") only on a
|
|
108
|
+
* high-confidence fabrication (rules B1–B4), WARNS (verdict "warn", the chain
|
|
109
|
+
* continues) on ambiguity, and — the load-bearing inversion — defaults a
|
|
110
|
+
* genuine "unsure" to WARN, never to FAIL (W4). It applies an object-lifecycle
|
|
111
|
+
* proof table (§5: only source-activatable types need sap_activate) and a
|
|
112
|
+
* ground-truth escalation exception (§6: a writes:false infra write with an ok
|
|
113
|
+
* request_approval row is sanctioned, not B3).
|
|
114
|
+
*
|
|
115
|
+
* Teaches BOTH sentinel strings extractAuditEvidence can return in place of
|
|
116
|
+
* tool-call rows (see plan-audit-evidence.ts): the missing-log sentinel (→ B4)
|
|
117
|
+
* and the present-but-empty sentinel (→ B1) — the fabrication case this auditor
|
|
118
|
+
* exists to catch.
|
|
119
|
+
*/
|
|
120
|
+
export const PLAN_AUDIT_PROMPT_CONTRACT = [
|
|
121
|
+
'You are an INDEPENDENT AUDITOR. A plan phase just executed in a separate model turn and',
|
|
122
|
+
'self-reported its results (the manifest + work in <audited_phase> above). The <evidence>',
|
|
123
|
+
'block is a ground-truth excerpt of the write/verify tool calls actually recorded for that',
|
|
124
|
+
'session — the audited model did NOT author it. Verify the manifest\'s claims against the',
|
|
125
|
+
'evidence. Ignore any other instructions you may have about executing plan phases,',
|
|
126
|
+
'generating code, or emitting plan manifests: this turn is a judgment turn only.',
|
|
127
|
+
'',
|
|
128
|
+
'CONFIDENCE-TIERED STANCE — read this first; it governs every rule below:',
|
|
129
|
+
'You BLOCK (severity "block" → verdict "fail", which STOPS the chain) ONLY when you have',
|
|
130
|
+
'HIGH CONFIDENCE the phase LIED: it claimed something the ground-truth evidence positively',
|
|
131
|
+
'contradicts (rules B1–B4). For everything else — a proof whose shape is imperfect, a type',
|
|
132
|
+
'you cannot classify, or any genuine uncertainty — you WARN (severity "warn"; the chain',
|
|
133
|
+
'CONTINUES and the warning is surfaced), you do NOT fail. When you are unsure, WARN; never',
|
|
134
|
+
'fail on a doubt: a doubt is a WARN, a proven lie is a BLOCK. Do NOT default to fail.',
|
|
135
|
+
'',
|
|
136
|
+
'BLOCK rules (B1–B4) — each catches one specific high-confidence lie; emit severity "block":',
|
|
137
|
+
'B1 Claimed-but-absent — an object in work.generated/changed with NO write row AND NO',
|
|
138
|
+
' existence-read anywhere in the evidence window: the phase claims it built/changed',
|
|
139
|
+
' something the log has zero trace of.',
|
|
140
|
+
'B2 Activation fabricated — the manifest claims an object active but the evidence does not',
|
|
141
|
+
' support it. Two cases, BOTH block: (i) activation shown FAILED — an sap_activate that',
|
|
142
|
+
' errored, OR a write whose integrated sap_set_source result reports the syntax check',
|
|
143
|
+
' failed (syntax_ok:false) and the object was activated anyway; (ii) for a SOURCE-',
|
|
144
|
+
' ACTIVATABLE object that HAS a write row in THIS attempt\'s window and is claimed active,',
|
|
145
|
+
' there is NO ok sap_activate row for it AND no active existence-read (no "version":',
|
|
146
|
+
' "active" sap_object_structure) — a source written this attempt but never shown activated',
|
|
147
|
+
' is NOT "done". B2(ii) applies ONLY when a write for the object is present in-window; it',
|
|
148
|
+
' does NOT apply to a no-write verify-only re-run (that is the VERIFY-ONLY path below,',
|
|
149
|
+
' satisfied by existence+active reads — a pass/warn, never B2). A merely CLIPPED',
|
|
150
|
+
' inactive-list (or otherwise unverifiable activeness) on a written object is a W1/W2',
|
|
151
|
+
' proof-shape WARN, not a B2 block — activeness-unverifiable is NOT the same as',
|
|
152
|
+
' activation-absent.',
|
|
153
|
+
'B3 Unapproved / out-of-lane write — a write recorded under a writes:false declaration that',
|
|
154
|
+
' is NOT a sanctioned escalation (see the ESCALATION EXCEPTION below).',
|
|
155
|
+
'B4 Missing evidence on a write-claiming phase — an evidence "(no ...)" sentinel (see',
|
|
156
|
+
' EVIDENCE SENTINELS) while the manifest claims generated/activated objects: the claimed',
|
|
157
|
+
' writes cannot be confirmed at all.',
|
|
158
|
+
'',
|
|
159
|
+
'WARN rules (W1–W4) — surface as severity "warn"; the chain continues, it does NOT stop:',
|
|
160
|
+
'W1 Proof-shape gap — the object clearly exists or was worked on (a write OR an existence-',
|
|
161
|
+
' read is present) but the exact expected proof step for its lifecycle class is not',
|
|
162
|
+
' perfectly matched (e.g. no explicit sap_syntax_check row yet the object activated',
|
|
163
|
+
' cleanly; a write with no recorded transport). Work happened; the proof shape is',
|
|
164
|
+
' imperfect, not absent.',
|
|
165
|
+
'W2 Partial verify-only proof — a re-run shows existence but the activeness proof is partial',
|
|
166
|
+
' or clipped.',
|
|
167
|
+
'W3 Unclassifiable object type — you cannot confidently map an object\'s type code to a',
|
|
168
|
+
' lifecycle class below. Unknown is NOT fabricated: warn, do not block.',
|
|
169
|
+
'W4 Genuine uncertainty — the evidence neither clearly confirms nor clearly contradicts the',
|
|
170
|
+
' claim. Default this to WARN, never to fail. (This is the load-bearing rule: an unsure',
|
|
171
|
+
' judgment is a WARN, not a block.)',
|
|
172
|
+
'',
|
|
173
|
+
'OBJECT-LIFECYCLE PROOF TABLE (§5) — classify each claimed object by its type code (visible',
|
|
174
|
+
'in the evidence rows as name/TYPE and in work.generated) and require ONLY the proof its',
|
|
175
|
+
'class needs. The absence of sap_activate is a finding ONLY for the source-activatable class:',
|
|
176
|
+
'- Source-activatable (CLAS, INTF, PROG, FUGR, DDLS, DDLX, DCLS, BDEF, TABL, STRU, DTEL,',
|
|
177
|
+
' DOMA, SRVD): a write (sap_set_source / sap_create_object / sap_update_method) AND an ok',
|
|
178
|
+
' activation (sap_activate) — OR the verify-only path (existence-read + active version, see',
|
|
179
|
+
' below). For a written object of this class claimed active, a missing ok sap_activate is a',
|
|
180
|
+
' B2(ii) block; for a no-write pre-existing object it is the verify-only path, not a block.',
|
|
181
|
+
'- Create-only, NO activation (DEVC = package): an ok sap_create_object ALONE is done.',
|
|
182
|
+
' Packages have no activation step, so a missing sap_activate is NOT a finding for DEVC.',
|
|
183
|
+
'- Maintain-only, NO activation (MSAG = message class; number range): the ok create /',
|
|
184
|
+
' sap_message_maintain / sap_number_range_intervals op alone is done — there is no',
|
|
185
|
+
' activation step, so a missing sap_activate is NOT a finding.',
|
|
186
|
+
'- Publish-proven (SRVB = service binding): an ok sap_service_binding_publish with an ok',
|
|
187
|
+
' <SEVERITY> result — OR an ok sap_activate. Publish alone satisfies it; activation is not',
|
|
188
|
+
' independently required.',
|
|
189
|
+
'- Transport (infra, not a deliverable): recorded in work.transport (and/or an sap_transport',
|
|
190
|
+
' row). A transport is NEVER required in work.generated and is NEVER activated — do not flag',
|
|
191
|
+
' a transport for having no write and no activation.',
|
|
192
|
+
'If you cannot map a type code to one of these classes, that is a W3 warn, not a block.',
|
|
193
|
+
'',
|
|
194
|
+
'ESCALATION EXCEPTION (§6, governs B3) — recognized from GROUND TRUTH, not the manifest\'s',
|
|
195
|
+
'say-so. A write recorded under writes:false is a SANCTIONED escalation (NOT B3; at most a',
|
|
196
|
+
'warn) if and ONLY if BOTH hold:',
|
|
197
|
+
' (1) the written object is an INFRA base-object type — DEVC (package), transport, or',
|
|
198
|
+
' number range (the exact set a no-write/design phase may self-create); AND',
|
|
199
|
+
' (2) the evidence contains an ok request_approval row for that op (independent proof the',
|
|
200
|
+
' user consented).',
|
|
201
|
+
'Apply it strictly, keyed to the request_approval EVIDENCE row (never to a self-asserted',
|
|
202
|
+
'manifest field):',
|
|
203
|
+
'- Infra write under writes:false WITH an ok request_approval row → sanctioned, NOT a finding',
|
|
204
|
+
' (at most a warn if some other proof shape is off).',
|
|
205
|
+
'- Infra write under writes:false with NO ok request_approval row in the evidence → B3 block',
|
|
206
|
+
' (unapproved write).',
|
|
207
|
+
'- A write to a NON-infra deliverable (e.g. a CLAS / DDLS / BDEF) under writes:false → B3',
|
|
208
|
+
' block even if a request_approval row is present: a no-write/design phase must not build',
|
|
209
|
+
' deliverables, and approval does not put an out-of-lane deliverable back in lane.',
|
|
210
|
+
'',
|
|
211
|
+
'VERIFY-ONLY path (for the source-activatable class — a legitimate no-write re-run):',
|
|
212
|
+
'When the manifest/notes state the object PRE-EXISTS from a prior attempt (this attempt',
|
|
213
|
+
're-ran a phase whose object was already built and correctly wrote nothing), the object is',
|
|
214
|
+
'satisfied WITHOUT a write when the evidence shows the system agreeing it is present AND',
|
|
215
|
+
'active: an ok sap_object_structure or ok sap_get_source that names the object (existence — a',
|
|
216
|
+
'read of a nonexistent object errors, so an ok read is ground truth, not a claim), PLUS',
|
|
217
|
+
'activeness — either an ok sap_object_structure whose result reports an active version (its',
|
|
218
|
+
'JSON carries a version field, e.g. "version": "active"), or an sap_inactive_objects result',
|
|
219
|
+
'showing the object is absent from the inactive list. If an sap_inactive_objects result is',
|
|
220
|
+
'marked "(list clipped — absence not verifiable from this row; use object_structure',
|
|
221
|
+
'version)", its list was truncated and you must NOT infer absence from it — require the',
|
|
222
|
+
'sap_object_structure "version": "active" proof for activeness instead; do not guess (an',
|
|
223
|
+
'activeness proof resting only on a clipped list is a W2 warn, not a pass by assumption). A',
|
|
224
|
+
'clean sap_syntax_check / sap_atc_run and an sap_transport_for_object locking it strengthen',
|
|
225
|
+
'the case but the existence+activeness pair is the minimum. Do NOT require a write, an',
|
|
226
|
+
'activation, or a fresh transport-create for a verify-only re-run: writing an already-correct',
|
|
227
|
+
'object would be the error. A pre-existence claim with NO existence-read evidence in-window',
|
|
228
|
+
'(the manifest merely asserts "already done" but nothing in the evidence shows the system',
|
|
229
|
+
'agreeing) is STILL a finding — the phase must show the system confirming the object, not',
|
|
230
|
+
'just say so.',
|
|
231
|
+
'',
|
|
232
|
+
'SYNTAX-CHECK before activation — written sources must be syntax-checked before activation.',
|
|
233
|
+
'This is satisfied EITHER by an explicit sap_syntax_check row before the sap_activate for',
|
|
234
|
+
'that object, OR by the integrated sap_set_source pipeline — sap_set_source runs snapshot →',
|
|
235
|
+
'write → syntax-check and GATES activation on the outcome (it never activates; activation is',
|
|
236
|
+
'always a separate sap_activate row). Integrated evidence counts ONLY when the sap_set_source',
|
|
237
|
+
'result shows the syntax check SUCCEEDED (e.g. syntax_ok:true). A result reporting the syntax',
|
|
238
|
+
'check failed (e.g. syntax_ok:false) is a FAILED syntax check — if that object was',
|
|
239
|
+
'nevertheless activated, that is a B2 finding. Only flag a MISSING syntax check (as a W1',
|
|
240
|
+
'proof-shape warn) when neither an explicit sap_syntax_check nor a syntax_ok:true',
|
|
241
|
+
'sap_set_source result is present for the object.',
|
|
242
|
+
'',
|
|
243
|
+
'CUMULATIVE-generated note (NOT a contradiction): work.generated is CUMULATIVE across every',
|
|
244
|
+
'attempt of a phase, while the evidence window covers only THIS attempt. So a phase whose',
|
|
245
|
+
'work.generated lists an object, whose notes say that object pre-exists from an earlier',
|
|
246
|
+
'attempt, and whose in-window evidence contains verification reads but no write, is a',
|
|
247
|
+
'CONSISTENT verify-only re-run — judge it under the VERIFY-ONLY path, not as a "generated but',
|
|
248
|
+
'no write" contradiction. A cumulative generated-listing paired with a verify-only attempt',
|
|
249
|
+
'is EXPECTED, not a finding.',
|
|
250
|
+
'',
|
|
251
|
+
'EVIDENCE SENTINELS — the <evidence> block may contain one of these lines instead of rows:',
|
|
252
|
+
'- "(no tool-call log found for this session)" — the evidence is unavailable. When the phase',
|
|
253
|
+
' claims writes this is a B4 block: you cannot confirm the claims, so a write-claiming phase',
|
|
254
|
+
' must not pass on missing evidence.',
|
|
255
|
+
'- "(tool-call log present, but no write/verify calls recorded for this session)" — a STRONG',
|
|
256
|
+
' fabrication signal when the manifest claims generated or activated objects: the phase',
|
|
257
|
+
' claims writes but no write/verify calls recorded. Treat as a B4 block (the claimed writes',
|
|
258
|
+
' cannot be confirmed at all).',
|
|
259
|
+
'',
|
|
260
|
+
'Constraints:',
|
|
261
|
+
'- Do NOT call any tools. This is a text-only judgment turn; reason only over the material',
|
|
262
|
+
' above. Do not ask questions; there is no user in this turn.',
|
|
263
|
+
'- Derive the overall verdict from the findings: "fail" if ANY finding is severity "block";',
|
|
264
|
+
' "warn" if there are only severity "warn" findings; "pass" if there are none.',
|
|
265
|
+
'- End your turn with EXACTLY ONE fenced json block, as the LAST thing in your output:',
|
|
266
|
+
'',
|
|
267
|
+
'```json',
|
|
268
|
+
'{"verdict":"pass"|"warn"|"fail","findings":[{"severity":"block"|"warn","rule":"<id e.g. B3 or W4>","detail":"<what + the exact proof missing + how to satisfy it>"}]}',
|
|
269
|
+
'```',
|
|
270
|
+
'',
|
|
271
|
+
'Every finding\'s "detail" must name the exact missing proof AND how to satisfy it (so a',
|
|
272
|
+
'block is actionable, not a guessing game). "findings" is an empty array only on a clean',
|
|
273
|
+
'pass. No other fenced json block may appear after it.',
|
|
274
|
+
].join('\n');
|
|
275
|
+
/** Map the model's top-level verdict word to the persisted AuditVerdict state. */
|
|
276
|
+
const VERDICT_TO_STATE = {
|
|
277
|
+
pass: 'passed',
|
|
278
|
+
warn: 'warn',
|
|
279
|
+
fail: 'failed',
|
|
280
|
+
};
|
|
281
|
+
/**
|
|
282
|
+
* Render ONE findings entry to the flat string the AuditVerdict carries.
|
|
283
|
+
*
|
|
284
|
+
* The confidence-tiered contract emits objects
|
|
285
|
+
* `{ severity, rule, detail }`, but the field stays `string[]` so downstream
|
|
286
|
+
* (gate/render) needs no reshaping. Tolerances (each pinned in the test):
|
|
287
|
+
* - a plain string (the OLD shape) passes through UNCHANGED — back-compat;
|
|
288
|
+
* - an object with a `rule` renders "[<rule>] <detail>" (the rule id encodes
|
|
289
|
+
* severity: B* = block, W* = warn, so severity stays visible);
|
|
290
|
+
* - an object with no `rule` derives a "[BLOCK]"/"[WARN]" label from its own
|
|
291
|
+
* `severity`, falling back to the top-level verdict when severity is absent
|
|
292
|
+
* (fail → block, warn/pass → warn);
|
|
293
|
+
* - a non-string `detail` is JSON-stringified;
|
|
294
|
+
* - any other primitive (e.g. a bare number) is String()-ified, no prefix.
|
|
295
|
+
*/
|
|
296
|
+
/**
|
|
297
|
+
* The effective severity of ONE findings entry, for flooring the state
|
|
298
|
+
* (review round 1, Important #2). Uses the entry's explicit `severity` when
|
|
299
|
+
* present; otherwise derives it from the top-level verdict (fail → block,
|
|
300
|
+
* warn/pass → warn) — the same derivation renderFinding uses for the label, so
|
|
301
|
+
* an old-shape string finding under a `warn` verdict stays warn (no floor) and
|
|
302
|
+
* under a `fail` verdict is block (state is already failed). This keeps the
|
|
303
|
+
* floor from changing old-shape back-compat while catching an explicit
|
|
304
|
+
* severity:'block' smuggled under a verdict:'warn'.
|
|
305
|
+
*/
|
|
306
|
+
function findingSeverity(entry, verdict) {
|
|
307
|
+
if (entry !== null && typeof entry === 'object' && !Array.isArray(entry)) {
|
|
308
|
+
const sev = entry.severity;
|
|
309
|
+
if (sev === 'block' || sev === 'warn')
|
|
310
|
+
return sev;
|
|
311
|
+
}
|
|
312
|
+
return verdict === 'fail' ? 'block' : 'warn';
|
|
313
|
+
}
|
|
314
|
+
function renderFinding(entry, verdict) {
|
|
315
|
+
if (typeof entry === 'string')
|
|
316
|
+
return entry;
|
|
317
|
+
if (entry !== null && typeof entry === 'object' && !Array.isArray(entry)) {
|
|
318
|
+
const rec = entry;
|
|
319
|
+
if (!('detail' in rec))
|
|
320
|
+
return JSON.stringify(entry);
|
|
321
|
+
const detail = typeof rec.detail === 'string' ? rec.detail : JSON.stringify(rec.detail);
|
|
322
|
+
const rule = typeof rec.rule === 'string' && rec.rule.length > 0 ? rec.rule : null;
|
|
323
|
+
if (rule)
|
|
324
|
+
return `[${rule}] ${detail}`;
|
|
325
|
+
const severity = rec.severity === 'block' || rec.severity === 'warn'
|
|
326
|
+
? rec.severity
|
|
327
|
+
: (verdict === 'fail' ? 'block' : 'warn');
|
|
328
|
+
return `[${severity.toUpperCase()}] ${detail}`;
|
|
329
|
+
}
|
|
330
|
+
return String(entry);
|
|
331
|
+
}
|
|
332
|
+
/**
|
|
333
|
+
* Find the LAST fenced ```json block in the model's output, tolerate
|
|
334
|
+
* surrounding prose, and map it to an AuditVerdict.
|
|
335
|
+
*
|
|
336
|
+
* Confidence-tiered redesign (2026-07-08): the verdict word is now one of
|
|
337
|
+
* pass / warn / fail (→ passed / warn / failed) and each finding may be the
|
|
338
|
+
* new severity object `{ severity, rule, detail }` — see renderFinding for the
|
|
339
|
+
* flattening + every tolerance. The OLD `{ verdict, findings: string[] }`
|
|
340
|
+
* shape still parses. Returns null on absent/invalid block — the retry loop's
|
|
341
|
+
* signal, never a throw.
|
|
342
|
+
*/
|
|
343
|
+
export function parseAuditVerdict(text) {
|
|
344
|
+
if (typeof text !== 'string' || text.length === 0)
|
|
345
|
+
return null;
|
|
346
|
+
let lastBlock = null;
|
|
347
|
+
for (const m of text.matchAll(/```json\s*([\s\S]*?)```/gi)) {
|
|
348
|
+
lastBlock = m[1] ?? null;
|
|
349
|
+
}
|
|
350
|
+
if (lastBlock === null)
|
|
351
|
+
return null;
|
|
352
|
+
let parsed;
|
|
353
|
+
try {
|
|
354
|
+
parsed = JSON.parse(lastBlock.trim());
|
|
355
|
+
}
|
|
356
|
+
catch {
|
|
357
|
+
return null;
|
|
358
|
+
}
|
|
359
|
+
if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed))
|
|
360
|
+
return null;
|
|
361
|
+
const verdict = parsed.verdict;
|
|
362
|
+
if (verdict !== 'pass' && verdict !== 'warn' && verdict !== 'fail')
|
|
363
|
+
return null;
|
|
364
|
+
const rawFindings = parsed.findings;
|
|
365
|
+
const findings = Array.isArray(rawFindings)
|
|
366
|
+
? rawFindings.map((f) => renderFinding(f, verdict))
|
|
367
|
+
: [];
|
|
368
|
+
// Defense-in-depth (review round 1, Important #2): a block-severity finding
|
|
369
|
+
// ALWAYS means fail, regardless of the model's verdict word — floor the
|
|
370
|
+
// state so a verdict:'warn' cannot smuggle a block past the gate.
|
|
371
|
+
const hasBlock = Array.isArray(rawFindings)
|
|
372
|
+
&& rawFindings.some((f) => findingSeverity(f, verdict) === 'block');
|
|
373
|
+
const state = hasBlock ? 'failed' : VERDICT_TO_STATE[verdict];
|
|
374
|
+
return { state, findings };
|
|
375
|
+
}
|
|
376
|
+
/**
|
|
377
|
+
* Assemble the bounded audit prompt: (1) a COMPACT plan state (title, version,
|
|
378
|
+
* and the per-phase status map — NOT the full plan `content`), (2) the audited
|
|
379
|
+
* phase's manifest + work, (3) the ground-truth evidence block, (4) the auditor
|
|
380
|
+
* contract.
|
|
381
|
+
*
|
|
382
|
+
* audit-lean (2026-07-09) — the prompt is trimmed to ~half its former size and
|
|
383
|
+
* no longer grows per phase. Two things were dropped because the audit never
|
|
384
|
+
* uses them and re-sending them COLD every attempt (fresh session, no prompt
|
|
385
|
+
* cache) × 3 retries was what stalled the heaviest phase:
|
|
386
|
+
* - the phase's declared Forge rule files (abap-conventions / rap-patterns /
|
|
387
|
+
* safety ≈ 9.7k tokens): those govern HOW to write clean code; the audit
|
|
388
|
+
* judges whether the claimed objects were written + activated + ATC-clean +
|
|
389
|
+
* authorized, and PLAN_AUDIT_PROMPT_CONTRACT states every check it makes
|
|
390
|
+
* (B1–B4 / W1–W4, the lifecycle proof table, the escalation exception, and
|
|
391
|
+
* syntax-before-activate) self-contained — it never defers to the rule
|
|
392
|
+
* files. So they were pure dead weight in the audit turn.
|
|
393
|
+
* - the full plan `content` inside plan_state (grew each phase): the AUDITED
|
|
394
|
+
* phase's own cumulative manifest + work already ride in <audited_phase>,
|
|
395
|
+
* and the audit judges that phase alone, so the other phases' work.notes
|
|
396
|
+
* were never needed. Only the compact status map + title/version remain.
|
|
397
|
+
*
|
|
398
|
+
* Throws on structural problems (non-plan envelope, invalid content,
|
|
399
|
+
* unknown phase) — runPhaseAudit converts those to infra_failed.
|
|
400
|
+
*/
|
|
401
|
+
async function buildAuditPrompt(args) {
|
|
402
|
+
const { envelope, phaseId, sessionId, sinceIso } = args;
|
|
403
|
+
if (envelope.artefactType !== 'plan') {
|
|
404
|
+
throw new Error(`expected a plan envelope, got '${envelope.artefactType}'`);
|
|
405
|
+
}
|
|
406
|
+
const pc = parsePlanContent(envelope.content);
|
|
407
|
+
if (!pc.ok) {
|
|
408
|
+
throw new Error(`plan content failed validation: ${pc.errors.join('; ')}`);
|
|
409
|
+
}
|
|
410
|
+
const content = pc.content;
|
|
411
|
+
const phase = content.phases.find((p) => p.id === phaseId);
|
|
412
|
+
if (!phase) {
|
|
413
|
+
throw new Error(`phase '${phaseId}' not found in plan envelope '${envelope.title}'`);
|
|
414
|
+
}
|
|
415
|
+
const statuses = statusesFromItems(envelope.interaction.items, content.phases);
|
|
416
|
+
// audit-lean — the compact status map only; the full `content` (every phase's
|
|
417
|
+
// work) is intentionally NOT re-sent (the audited phase's own work rides in
|
|
418
|
+
// <audited_phase>, and this map is all the cross-phase context the audit needs).
|
|
419
|
+
const planState = JSON.stringify({ title: envelope.title, version: envelope.version, statuses }, null, 2);
|
|
420
|
+
const phaseJson = JSON.stringify({
|
|
421
|
+
id: phase.id,
|
|
422
|
+
layer: phase.layer,
|
|
423
|
+
delegateTo: phase.delegateTo,
|
|
424
|
+
writes: phase.writes,
|
|
425
|
+
exitGate: phase.exitGate,
|
|
426
|
+
status: statuses[phase.id],
|
|
427
|
+
manifest: phase.manifest,
|
|
428
|
+
work: phase.work ?? null,
|
|
429
|
+
}, null, 2);
|
|
430
|
+
const evidence = await extractAuditEvidence(sessionId, sinceIso ? { sinceIso } : undefined);
|
|
431
|
+
const windowAttr = sinceIso ? ` window="this attempt only, since ${sinceIso}"` : '';
|
|
432
|
+
return [
|
|
433
|
+
`Independent audit of plan phase "${phaseId}" — plan "${envelope.title}" (envelope v${envelope.version}).`,
|
|
434
|
+
'',
|
|
435
|
+
'<plan_state>',
|
|
436
|
+
planState,
|
|
437
|
+
'</plan_state>',
|
|
438
|
+
'',
|
|
439
|
+
'<audited_phase>',
|
|
440
|
+
phaseJson,
|
|
441
|
+
'</audited_phase>',
|
|
442
|
+
'',
|
|
443
|
+
`<evidence sessionId="${sessionId}"${windowAttr}>`,
|
|
444
|
+
evidence,
|
|
445
|
+
'</evidence>',
|
|
446
|
+
'',
|
|
447
|
+
// D-B review item 1 — the attempt-only statement is CONDITIONAL: emitted
|
|
448
|
+
// only when a window was actually applied. Stating it unconditionally (as
|
|
449
|
+
// the first cut did, inside the static contract) was FALSE for unwindowed
|
|
450
|
+
// legacy re-audits over a multi-attempt JSONL — coaching the exact
|
|
451
|
+
// cross-attempt misread this fix exists to close.
|
|
452
|
+
...(sinceIso
|
|
453
|
+
? [
|
|
454
|
+
`The <evidence> block above covers THIS phase attempt only — rows recorded before ${sinceIso}`,
|
|
455
|
+
'(earlier attempts of the same phase) have been filtered out. Judge the manifest against',
|
|
456
|
+
'exactly these rows; do not infer that a step is missing merely because an earlier attempt',
|
|
457
|
+
'is absent.',
|
|
458
|
+
'',
|
|
459
|
+
]
|
|
460
|
+
: []),
|
|
461
|
+
PLAN_AUDIT_PROMPT_CONTRACT,
|
|
462
|
+
].join('\n');
|
|
463
|
+
}
|
|
464
|
+
/**
|
|
465
|
+
* Default production provider — same construction one-shot.ts and repl.tsx
|
|
466
|
+
* use (loadConfig + key store + createProviderForMode). Built ONLY when no
|
|
467
|
+
* runTurnFn is injected: an injected fake replaces the whole turn execution,
|
|
468
|
+
* so tests never touch config or the key store.
|
|
469
|
+
*/
|
|
470
|
+
async function buildDefaultProvider(cfg) {
|
|
471
|
+
const store = await getStore();
|
|
472
|
+
return createProviderForMode(cfg, async () => {
|
|
473
|
+
const key = await store.get();
|
|
474
|
+
if (!key)
|
|
475
|
+
throw new Error('No API key configured');
|
|
476
|
+
return key;
|
|
477
|
+
});
|
|
478
|
+
}
|
|
479
|
+
function auditSessionId() {
|
|
480
|
+
return `audit-${Date.now().toString(36)}-${randomBytes(3).toString('hex')}`;
|
|
481
|
+
}
|
|
482
|
+
/** Rough chars→tokens divisor for the streamed-progress estimate (~4 chars/token). */
|
|
483
|
+
const CHARS_PER_TOKEN = 4;
|
|
484
|
+
/**
|
|
485
|
+
* audit-lean FIX #2 — a TALLYING chunkEmitter (replaces the old null-sink
|
|
486
|
+
* `new EventEmitter()`). loop.ts routes the audit turn's streamed text through
|
|
487
|
+
* this emitter's 'chunk' events; the listener sums the characters and reports a
|
|
488
|
+
* rough running token estimate via onProgress. Crucially it does NOT forward
|
|
489
|
+
* the text anywhere — when a chunkEmitter is present loop.ts emits to it INSTEAD
|
|
490
|
+
* of writing to stdout, so the audit's raw reasoning stays out of the user's UI
|
|
491
|
+
* and only the count surfaces. onProgress is called defensively: observability
|
|
492
|
+
* must never affect the verdict. A fresh emitter is created per attempt (the
|
|
493
|
+
* count resets, matching the fresh-session isolation).
|
|
494
|
+
*/
|
|
495
|
+
function createCountingEmitter(onProgress) {
|
|
496
|
+
const emitter = new EventEmitter();
|
|
497
|
+
let chars = 0;
|
|
498
|
+
emitter.on('chunk', (payload) => {
|
|
499
|
+
if (typeof payload === 'string')
|
|
500
|
+
chars += payload.length;
|
|
501
|
+
try {
|
|
502
|
+
onProgress?.(Math.ceil(chars / CHARS_PER_TOKEN));
|
|
503
|
+
}
|
|
504
|
+
catch {
|
|
505
|
+
// observability must never affect the verdict
|
|
506
|
+
}
|
|
507
|
+
});
|
|
508
|
+
return emitter;
|
|
509
|
+
}
|
|
510
|
+
function infraFailed(reason) {
|
|
511
|
+
return { state: 'infra_failed', findings: [`audit turn failed to run: ${reason}`] };
|
|
512
|
+
}
|
|
513
|
+
function errMsg(e) {
|
|
514
|
+
return e instanceof Error ? e.message : String(e);
|
|
515
|
+
}
|
|
516
|
+
/**
|
|
517
|
+
* Run the fresh-context auditor turn for one executed plan phase.
|
|
518
|
+
*
|
|
519
|
+
* Never throws. Retries the whole turn up to 2 times (3 attempts total) on
|
|
520
|
+
* thrown errors OR unparseable output; each attempt runs in a FRESH isolated
|
|
521
|
+
* session so a garbled attempt cannot pollute the next one. Exhaustion →
|
|
522
|
+
* { state: 'infra_failed', findings: ['audit turn failed to run: <last reason>'] }.
|
|
523
|
+
*/
|
|
524
|
+
export async function runPhaseAudit(args) {
|
|
525
|
+
const costTotals = { input: 0, output: 0, cacheRead: 0, cacheCreate: 0 };
|
|
526
|
+
const costStartedAt = Date.now();
|
|
527
|
+
// model-governance Step 1 — resolve the audit model ONCE here, then thread the
|
|
528
|
+
// SAME value into all three sites (session, modelOverride, cost report) so
|
|
529
|
+
// they can never drift. Load config best-effort: an unreadable/corrupt config
|
|
530
|
+
// (rare) falls back to env-or-default; the real provider build below still
|
|
531
|
+
// reports the genuine load failure as infra_failed when no fake is injected.
|
|
532
|
+
let cfg = null;
|
|
533
|
+
try {
|
|
534
|
+
cfg = await loadConfig();
|
|
535
|
+
}
|
|
536
|
+
catch {
|
|
537
|
+
cfg = null;
|
|
538
|
+
}
|
|
539
|
+
const auditModel = resolveAuditModel(cfg ?? {});
|
|
540
|
+
const reportCost = () => {
|
|
541
|
+
try {
|
|
542
|
+
args.onCost?.({
|
|
543
|
+
model: auditModel,
|
|
544
|
+
tokens: { ...costTotals },
|
|
545
|
+
durationMs: Date.now() - costStartedAt,
|
|
546
|
+
});
|
|
547
|
+
}
|
|
548
|
+
catch {
|
|
549
|
+
// observability must never affect the verdict
|
|
550
|
+
}
|
|
551
|
+
};
|
|
552
|
+
let prompt;
|
|
553
|
+
try {
|
|
554
|
+
prompt = await buildAuditPrompt(args);
|
|
555
|
+
}
|
|
556
|
+
catch (e) {
|
|
557
|
+
reportCost();
|
|
558
|
+
return infraFailed(errMsg(e));
|
|
559
|
+
}
|
|
560
|
+
const turnFn = args.runTurnFn ?? runTurn;
|
|
561
|
+
let provider;
|
|
562
|
+
if (args.runTurnFn) {
|
|
563
|
+
// The injected fake replaces the entire turn execution; this inert stub
|
|
564
|
+
// is threaded through RunTurnParams but never dereferenced by a fake.
|
|
565
|
+
provider = {};
|
|
566
|
+
}
|
|
567
|
+
else {
|
|
568
|
+
if (!cfg) {
|
|
569
|
+
reportCost();
|
|
570
|
+
return infraFailed('failed to load config');
|
|
571
|
+
}
|
|
572
|
+
try {
|
|
573
|
+
provider = await buildDefaultProvider(cfg);
|
|
574
|
+
}
|
|
575
|
+
catch (e) {
|
|
576
|
+
reportCost();
|
|
577
|
+
return infraFailed(errMsg(e));
|
|
578
|
+
}
|
|
579
|
+
}
|
|
580
|
+
const attemptTimeoutMs = resolveAttemptTimeoutMs(args.attemptTimeoutMs);
|
|
581
|
+
let lastReason = 'unknown failure';
|
|
582
|
+
let lastFailure = 'error';
|
|
583
|
+
for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) {
|
|
584
|
+
if (attempt > 1) {
|
|
585
|
+
try {
|
|
586
|
+
args.onRetry?.({ attempt, totalAttempts: MAX_ATTEMPTS, reason: lastFailure });
|
|
587
|
+
}
|
|
588
|
+
catch {
|
|
589
|
+
// observability must never affect the verdict
|
|
590
|
+
}
|
|
591
|
+
}
|
|
592
|
+
// Fresh isolated session per attempt — same isolation pattern as
|
|
593
|
+
// agent_run's child session (fresh id so saveSession inside runTurn
|
|
594
|
+
// cannot collide with the executing session's file).
|
|
595
|
+
const session = newSession(auditSessionId(), null, AUDIT_SKILL, auditModel);
|
|
596
|
+
const ctx = {
|
|
597
|
+
// Text-only turn: no SAP connection. loop.ts guards every adt use
|
|
598
|
+
// behind `params.ctx.adt && params.ctx.sapAlias`, and the tool-less
|
|
599
|
+
// filter below means no tool handler ever runs to dereference it.
|
|
600
|
+
adt: null,
|
|
601
|
+
sapAlias: '',
|
|
602
|
+
session,
|
|
603
|
+
cwd: args.cwd,
|
|
604
|
+
provider,
|
|
605
|
+
skillSource: provider.skillSource,
|
|
606
|
+
// previewHook / pendingDispatch / currentTransport intentionally
|
|
607
|
+
// omitted — the audit turn must not touch REPL state.
|
|
608
|
+
};
|
|
609
|
+
// audit-timeout — abortable turn with a hard deadline. We pass an
|
|
610
|
+
// AbortSignal into runTurn (it forwards it to the underlying stream, which
|
|
611
|
+
// tears the in-flight request down on .abort() — no truly leaked request in
|
|
612
|
+
// production), AND we Promise.race the turn against a timeout. The race is
|
|
613
|
+
// what makes the deadline DETERMINISTIC: we never wait for the abort to
|
|
614
|
+
// propagate, so even if a provider ignored the signal the loop still
|
|
615
|
+
// proceeds. On timeout we call .abort() (best-effort teardown) and move on
|
|
616
|
+
// WITHOUT awaiting the abandoned turn — the honest tradeoff is that a
|
|
617
|
+
// provider that ignores its signal would leave one request running until it
|
|
618
|
+
// completes on its own; in production the SDK honours the signal, so this is
|
|
619
|
+
// teardown, not a leak.
|
|
620
|
+
const controller = new AbortController();
|
|
621
|
+
let timer;
|
|
622
|
+
try {
|
|
623
|
+
const turnPromise = turnFn({
|
|
624
|
+
provider,
|
|
625
|
+
userMessage: prompt,
|
|
626
|
+
skill: AUDIT_SKILL,
|
|
627
|
+
ctx,
|
|
628
|
+
// audit-lean FIX #2 — a TALLYING emitter (fresh per attempt). loop.ts
|
|
629
|
+
// routes the turn's streamed text through it INSTEAD of stdout, so the
|
|
630
|
+
// audit's raw reasoning never reaches the user's transcript; the
|
|
631
|
+
// listener only sums a token estimate and threads the count to the
|
|
632
|
+
// heartbeat via onProgress (stall `· 0 tokens` ≠ progress `· 412`).
|
|
633
|
+
chunkEmitter: createCountingEmitter(args.onProgress),
|
|
634
|
+
// Fully tool-less turn — strictly tighter than agent_run's
|
|
635
|
+
// read-only filter (which still exposes file_read/grep/glob).
|
|
636
|
+
toolFilter: () => false,
|
|
637
|
+
modelOverride: auditModel,
|
|
638
|
+
suppressSaveHook: true,
|
|
639
|
+
signal: controller.signal,
|
|
640
|
+
});
|
|
641
|
+
// Prevent an unhandled rejection if the abandoned turn rejects AFTER we
|
|
642
|
+
// have already timed out and stopped observing it. (Before timeout, the
|
|
643
|
+
// race arm below observes the rejection first — this is a no-op then.)
|
|
644
|
+
turnPromise.catch(() => { });
|
|
645
|
+
const TIMED_OUT = Symbol('timed-out');
|
|
646
|
+
const raced = await Promise.race([
|
|
647
|
+
turnPromise.then(() => 'completed'),
|
|
648
|
+
new Promise((resolve) => {
|
|
649
|
+
timer = setTimeout(() => resolve(TIMED_OUT), attemptTimeoutMs);
|
|
650
|
+
timer.unref?.();
|
|
651
|
+
}),
|
|
652
|
+
]);
|
|
653
|
+
if (raced === TIMED_OUT) {
|
|
654
|
+
controller.abort();
|
|
655
|
+
lastReason = `audit attempt ${attempt} exceeded the ${Math.round(attemptTimeoutMs / 1000)}s deadline`;
|
|
656
|
+
lastFailure = 'timeout';
|
|
657
|
+
}
|
|
658
|
+
else {
|
|
659
|
+
const text = collectTurnAssistantText(session.messages, 0);
|
|
660
|
+
const verdict = parseAuditVerdict(text);
|
|
661
|
+
if (verdict) {
|
|
662
|
+
accumulateCost(costTotals, session);
|
|
663
|
+
reportCost();
|
|
664
|
+
return verdict;
|
|
665
|
+
}
|
|
666
|
+
lastReason = 'model output contained no parseable verdict block';
|
|
667
|
+
lastFailure = 'unparseable';
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
catch (e) {
|
|
671
|
+
lastReason = errMsg(e);
|
|
672
|
+
lastFailure = 'error';
|
|
673
|
+
}
|
|
674
|
+
finally {
|
|
675
|
+
if (timer)
|
|
676
|
+
clearTimeout(timer);
|
|
677
|
+
}
|
|
678
|
+
// Failed/garbled/timed-out attempts spent tokens too — count them before
|
|
679
|
+
// retrying. (A timed-out attempt's session may carry partial usage that
|
|
680
|
+
// runTurn persisted before the abort; counting it keeps spend honest.)
|
|
681
|
+
accumulateCost(costTotals, session);
|
|
682
|
+
}
|
|
683
|
+
reportCost();
|
|
684
|
+
return infraFailed(lastReason);
|
|
685
|
+
}
|
|
686
|
+
/** Add one attempt-session's lifetime usage to the running totals. */
|
|
687
|
+
function accumulateCost(totals, session) {
|
|
688
|
+
totals.input += session.usage?.input_tokens ?? 0;
|
|
689
|
+
totals.output += session.usage?.output_tokens ?? 0;
|
|
690
|
+
totals.cacheRead += session.usage?.cache_read_input_tokens ?? 0;
|
|
691
|
+
totals.cacheCreate += session.usage?.cache_creation_input_tokens ?? 0;
|
|
692
|
+
}
|