@cspeach/cli 0.9.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/agent/intent-system-prompt.js +1 -1
- package/dist/agent/loop.js +228 -26
- package/dist/agent/providers/license-gate.js +44 -0
- package/dist/agent/skill-checkpoint.js +1 -1
- package/dist/agent/tool-dispatch.js +15 -0
- package/dist/approvals/canonical.js +91 -0
- package/dist/approvals/jwt.js +39 -2
- package/dist/approvals/op-labels.js +124 -0
- package/dist/approvals/render.js +42 -36
- package/dist/auth/org-anthropic-key.js +25 -0
- package/dist/classifier/client.js +18 -3
- package/dist/cli.js +15 -0
- package/dist/commands/compact.js +28 -2
- package/dist/commands/config-set.js +284 -0
- package/dist/commands/config-show.js +20 -0
- package/dist/commands/export-audit.js +43 -0
- package/dist/commands/help.js +5 -0
- package/dist/commands/login.js +31 -14
- package/dist/commands/plan-audit-evidence.js +266 -0
- package/dist/commands/plan-audit.js +692 -0
- package/dist/commands/plan-chain.js +671 -0
- package/dist/commands/plan-continue.js +179 -0
- package/dist/commands/plan-gate.js +154 -0
- package/dist/commands/plan-model-tier.js +83 -0
- package/dist/commands/plan-resume.js +728 -46
- package/dist/config/loader.js +223 -5
- package/dist/config/model-defaults.js +14 -0
- package/dist/cost/pricing.js +27 -1
- package/dist/doctor/checks/_http-probe.js +1 -0
- package/dist/doctor/checks/cert.js +14 -3
- package/dist/doctor/checks/sap.js +30 -8
- package/dist/doctor/checks/system-roles.js +41 -0
- package/dist/doctor/checks/zcspeach.js +19 -4
- package/dist/doctor/run.js +2 -0
- package/dist/models/resolve.js +61 -0
- package/dist/models/server-config.js +155 -0
- package/dist/one-shot.js +76 -6
- package/dist/projects/answer-blockers.js +137 -0
- package/dist/projects/extract-cca.js +111 -17
- package/dist/projects/extract-modernize.js +4 -2
- package/dist/projects/extract-plan.js +184 -37
- package/dist/projects/extract-spec-gap.js +34 -7
- package/dist/projects/extract-test-coverage.js +4 -2
- package/dist/projects/extract-upgrade.js +116 -23
- package/dist/projects/handover-md.js +195 -0
- package/dist/projects/index.js +5 -2
- package/dist/projects/merge-cca.js +292 -0
- package/dist/projects/merge-upgrade.js +173 -0
- package/dist/projects/migration.js +103 -1
- package/dist/projects/output-paths.js +27 -0
- package/dist/projects/plan-run.js +285 -27
- package/dist/projects/plan-schema.js +136 -3
- package/dist/projects/promote-command.js +25 -2
- package/dist/projects/promote.js +128 -0
- package/dist/projects/run-lease.js +157 -0
- package/dist/projects/save-command.js +259 -21
- package/dist/projects/status.js +3 -1
- package/dist/projects/validate.js +1 -1
- package/dist/projects/workspace.js +164 -20
- package/dist/renderer/notices.js +64 -0
- package/dist/renderer/progress-chatter.js +8 -0
- package/dist/renderer/status-footer.js +22 -12
- package/dist/renderer/thinking-heartbeat.js +64 -8
- package/dist/renderer/todo-block.js +51 -0
- package/dist/renderer/tool-widget.js +55 -4
- package/dist/renderer/tty.js +43 -4
- package/dist/renderer/verify-chain.js +77 -0
- package/dist/repl/at-picker.js +60 -7
- package/dist/repl/bracketed-paste.js +28 -19
- package/dist/repl/builtin-commands.js +42 -0
- package/dist/repl/current-transport.js +10 -0
- package/dist/repl/early-line-buffer.js +68 -0
- package/dist/repl/history.js +86 -0
- package/dist/repl/ink-stdin-guard.js +64 -0
- package/dist/repl/inquirer-guard.js +70 -5
- package/dist/repl/mode-ceiling.js +16 -0
- package/dist/repl/mode-cycle.js +104 -0
- package/dist/repl/numbered-menu.js +131 -0
- package/dist/repl/post-turn-status.js +26 -6
- package/dist/repl/rule8-detector.js +17 -2
- package/dist/repl/safety-confirm.js +111 -2
- package/dist/repl/safety-mode-state.js +19 -3
- package/dist/repl/slash-completer.js +5 -0
- package/dist/repl/slash-picker.js +10 -15
- package/dist/repl.js +1232 -95
- package/dist/rewind/candidates.js +194 -0
- package/dist/rewind/cli.js +137 -0
- package/dist/rewind/format.js +27 -0
- package/dist/rewind/restore.js +245 -0
- package/dist/router/classifier.js +150 -6
- package/dist/sap/capability-matrix.js +20 -0
- package/dist/sap/capability-matrix.json +11236 -0
- package/dist/sap/capability.js +146 -0
- package/dist/sap/connection-manager.js +19 -1
- package/dist/sap/onboarding.js +42 -4
- package/dist/session/audit-export.js +459 -0
- package/dist/session/context-report.js +163 -0
- package/dist/session/pending.js +27 -0
- package/dist/session/recap.js +160 -0
- package/dist/skill-catalog.js +51 -40
- package/dist/skills/bundled-skills.js +272 -1
- package/dist/skills/promotion-dispatch.js +23 -0
- package/dist/tools/_command-shared.js +36 -12
- package/dist/tools/_filesystem-shared.js +139 -4
- package/dist/tools/_flag.js +25 -0
- package/dist/tools/approval.js +177 -26
- package/dist/tools/ask-question.js +400 -7
- package/dist/tools/capability/tool.js +74 -0
- package/dist/tools/dispatch-skill.js +22 -1
- package/dist/tools/extend-model/anchored-insert.js +1414 -0
- package/dist/tools/extend-model/tool.js +340 -0
- package/dist/tools/filesystem/extract-document.js +57 -0
- package/dist/tools/filesystem/file-edit.js +12 -2
- package/dist/tools/filesystem/file-read.js +2 -2
- package/dist/tools/filesystem/file-write.js +11 -2
- package/dist/tools/filesystem/glob.js +11 -0
- package/dist/tools/filesystem/grep.js +10 -0
- package/dist/tools/filesystem/read-document.js +107 -0
- package/dist/tools/fiori/apply.js +50 -0
- package/dist/tools/fiori/bin.js +3 -0
- package/dist/tools/fiori/catalog/index.js +27 -0
- package/dist/tools/fiori/catalog/value-help.js +230 -0
- package/dist/tools/fiori/catalog/viz-chart.js +177 -0
- package/dist/tools/fiori/cli.js +71 -0
- package/dist/tools/fiori/deploy-config.js +73 -0
- package/dist/tools/fiori/fe-extend.js +76 -0
- package/dist/tools/fiori/fe-scaffold.js +71 -0
- package/dist/tools/fiori/floorplan-map.js +19 -0
- package/dist/tools/fiori/i18n.js +39 -0
- package/dist/tools/fiori/manifest.js +70 -0
- package/dist/tools/fiori/render.js +77 -0
- package/dist/tools/fiori/samples/data/index.json +13602 -0
- package/dist/tools/fiori/samples/data/sources.generated.js +808 -0
- package/dist/tools/fiori/samples/loader.js +248 -0
- package/dist/tools/fiori/samples/search.js +63 -0
- package/dist/tools/fiori/samples/types.js +2 -0
- package/dist/tools/fiori/scaffold.js +39 -0
- package/dist/tools/fiori/smoke/assertions.js +74 -0
- package/dist/tools/fiori/smoke/browser.js +52 -0
- package/dist/tools/fiori/smoke/driver.js +89 -0
- package/dist/tools/fiori/smoke/freestyle-spec.js +317 -0
- package/dist/tools/fiori/smoke/run-smoke.js +149 -0
- package/dist/tools/fiori/tools.js +681 -0
- package/dist/tools/fiori/types.js +1 -0
- package/dist/tools/local-build.js +86 -0
- package/dist/tools/local-files.js +31 -0
- package/dist/tools/project/_merge-shared.js +68 -0
- package/dist/tools/project/cca_merge.js +164 -0
- package/dist/tools/project/playbook_get.js +1 -1
- package/dist/tools/project/upgrade_merge_progress.js +206 -0
- package/dist/tools/sap-read.js +132 -20
- package/dist/tools/sap-write.js +550 -21
- package/dist/tools/shell/shell_exec.js +41 -6
- package/dist/tools/snapshot.js +63 -14
- package/dist/tools/subagent/agent_run.js +27 -3
- package/dist/tools/subagent/background_run.js +17 -1
- package/dist/tools/todo.js +144 -0
- package/dist/tools/transport-resolution.js +86 -0
- package/dist/tools/transport.js +224 -5
- package/dist/tools/write-mode.js +4 -0
- package/dist/ui/app.js +378 -21
- package/dist/ui/approval-modal.js +49 -16
- package/dist/ui/ask-question-emitter.js +14 -0
- package/dist/ui/body.js +13 -0
- package/dist/ui/context-grid.js +108 -0
- package/dist/ui/footer.js +120 -27
- package/dist/ui/header.js +7 -0
- package/dist/ui/line-resolution.js +35 -8
- package/dist/ui/rewind-emitter.js +10 -0
- package/dist/ui/rewind-panel.js +81 -0
- package/dist/ui/sap-state-store.js +1 -0
- package/dist/ui/session-timeline.js +1 -0
- package/dist/ui/status-line.js +43 -0
- package/dist/ui/text-input.js +214 -0
- package/dist/ui/todo-emitter.js +25 -0
- package/dist/ui/todo-panel.js +64 -0
- package/dist/ui/turn-status-emitter.js +50 -4
- package/dist/ui/turn-status.js +18 -3
- package/dist/ui/widgets/ask-form.js +242 -0
- package/dist/ui/widgets/ask-question-modal.js +21 -8
- package/package.json +22 -3
- package/bench/README.md +0 -78
- package/bench/prompts/abap-document-cds.md +0 -44
- package/bench/prompts/abap-explain-bdef-handler.md +0 -57
- package/bench/prompts/abap-test-method.md +0 -42
- package/bench/results/abap-document-cds/claude-haiku-4-5.md +0 -189
- package/bench/results/abap-document-cds/claude-opus-4-7.md +0 -120
- package/bench/results/abap-document-cds/claude-sonnet-4-6.md +0 -151
- package/bench/results/abap-explain-bdef-handler/claude-haiku-4-5.md +0 -112
- package/bench/results/abap-explain-bdef-handler/claude-opus-4-7.md +0 -101
- package/bench/results/abap-explain-bdef-handler/claude-sonnet-4-6.md +0 -101
- package/bench/results/abap-test-method/claude-haiku-4-5.md +0 -186
- package/bench/results/abap-test-method/claude-opus-4-7.md +0 -193
- package/bench/results/abap-test-method/claude-sonnet-4-6.md +0 -234
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Ground-truth extractor for the phase auditor.
|
|
3
|
+
*
|
|
4
|
+
* The agent loop writes a JSONL audit log of every tool call to
|
|
5
|
+
* `~/.cspeach/checkpoints/<sessionId>/tool-calls.jsonl` (Layer 2 default
|
|
6
|
+
* handler in `agent/skill-checkpoint.ts`). The phase auditor must judge a
|
|
7
|
+
* phase's self-report against evidence the audited model did NOT author —
|
|
8
|
+
* this module produces that evidence: a bounded, compact excerpt of the
|
|
9
|
+
* write + verification tool calls (set_source, activate, syntax_check,
|
|
10
|
+
* transport, snapshot, ...), newest-last.
|
|
11
|
+
*
|
|
12
|
+
* Pure read module — no writes, no new dependencies.
|
|
13
|
+
*/
|
|
14
|
+
import { promises as fs } from 'node:fs';
|
|
15
|
+
import * as path from 'node:path';
|
|
16
|
+
import { checkpointsRoot } from '../agent/skill-checkpoint.js';
|
|
17
|
+
/**
|
|
18
|
+
* PROTECTED evidence — write + verification calls that PROVE a phase's core
|
|
19
|
+
* claims and must NEVER be evicted by read rows (see the two-tier cap in
|
|
20
|
+
* extractAuditEvidence). Transport tools are matched by their EXACT
|
|
21
|
+
* write-relevant names (fix-wave): the registered family is
|
|
22
|
+
* sap_transport_create / _for_object / _list / _release (tools/transport.ts),
|
|
23
|
+
* and a bare `sap_transport` prefix let list-shaped read polling
|
|
24
|
+
* (sap_transport_list, sap_transport_for_object) evict real write rows from
|
|
25
|
+
* the 60-row cap.
|
|
26
|
+
*
|
|
27
|
+
* D-A (2026-07-05): sap_atc_run and sap_inactive_objects are VERIFICATION
|
|
28
|
+
* tools the audit contract + phase exit gates explicitly demand (ATC clean /
|
|
29
|
+
* no inactive versions remain), yet they were missing from this filter — so
|
|
30
|
+
* a phase that legitimately ran ATC or checked for inactive objects had that
|
|
31
|
+
* evidence stripped, and the auditor false-failed it with "no ATC call
|
|
32
|
+
* appears anywhere in the evidence". They are reads in the sense of not
|
|
33
|
+
* mutating source, but they ARE the proof of the verify steps, so they
|
|
34
|
+
* belong in the evidence excerpt. (Unlike the transport LIST polling, these
|
|
35
|
+
* fire once per verify, not in a loop, so they do not threaten the row cap.)
|
|
36
|
+
*
|
|
37
|
+
* Task 2 (audit-confidence-tiered-redesign, §6): request_approval is admitted
|
|
38
|
+
* here too. It is the PROOF-of-consent the §6 escalation exception keys on — a
|
|
39
|
+
* write to an infra base-object (DEVC / transport / number-range) under a
|
|
40
|
+
* writes:false phase is a sanctioned escalation ONLY IF the evidence carries an
|
|
41
|
+
* OK request_approval row for it. With the old sap_*-only filter that row was
|
|
42
|
+
* dropped, so the exception could never be satisfied. Like the existence-read
|
|
43
|
+
* proofs it is ground truth (the user's recorded consent), and the whole
|
|
44
|
+
* contract branch depends on it — so it must NEVER be evicted by the row cap:
|
|
45
|
+
* PROTECTED, not the fillable existence tier. Matched EXACTLY (^…$ anchored) so
|
|
46
|
+
* a near-miss tool name can never masquerade as consent.
|
|
47
|
+
*/
|
|
48
|
+
const PROTECTED_TOOL_RE = /^sap_(set_source|update_method|create_object|delete_object|activate|syntax_check|atc_run|inactive_objects|transport_create|transport_release|service_binding_publish|snapshot)|^request_approval$/;
|
|
49
|
+
/**
|
|
50
|
+
* D-F (2026-07-05): EXISTENCE / VERIFY-READ evidence. A verify-only phase
|
|
51
|
+
* re-run (object already built in a prior attempt; this attempt proves it
|
|
52
|
+
* exists / is active / is clean and writes NOTHING) records only reads:
|
|
53
|
+
* sap_object_structure (version=active), sap_get_source (fields match), and
|
|
54
|
+
* sap_transport_for_object (object locked in a transport). With the old
|
|
55
|
+
* write-only filter those rows were stripped, the evidence went empty, and
|
|
56
|
+
* the auditor false-FAILED the re-run ("no write call in evidence"). These
|
|
57
|
+
* reads are legitimate ground truth: a read against the live harness log
|
|
58
|
+
* cannot be fabricated — a read of a nonexistent object ERRORS, so an `ok`
|
|
59
|
+
* existence read IS the system agreeing the object is there. They are
|
|
60
|
+
* admitted here so the auditor can satisfy a pre-existence claim on
|
|
61
|
+
* verification evidence alone (contract Rule 1, D-F branch).
|
|
62
|
+
*
|
|
63
|
+
* Cap safety (two-tier, see extractAuditEvidence): these reads are also used
|
|
64
|
+
* heavily in analysis phases and could, in a single-bucket cap, evict real
|
|
65
|
+
* write rows. They are therefore the FILLABLE tier — the newest of them fill
|
|
66
|
+
* whatever budget remains after every PROTECTED row is kept; a write row is
|
|
67
|
+
* never dropped to make room for one of these.
|
|
68
|
+
*/
|
|
69
|
+
const EXISTENCE_READ_RE = /^sap_(object_structure|get_source|transport_for_object)/;
|
|
70
|
+
/** Any row the auditor may see is the union of the two tiers. */
|
|
71
|
+
const EVIDENCE_TOOL_RE = new RegExp(`${PROTECTED_TOOL_RE.source}|${EXISTENCE_READ_RE.source}`);
|
|
72
|
+
const NO_LOG_SENTINEL = '(no tool-call log found for this session)';
|
|
73
|
+
const NO_EVIDENCE_ROWS = '(tool-call log present, but no write/verify calls recorded for this session)';
|
|
74
|
+
const DEFAULT_MAX_ROWS = 60;
|
|
75
|
+
const RESULT_EXCERPT_CHARS = 200;
|
|
76
|
+
const ARGS_COMPACT_CHARS = 80;
|
|
77
|
+
/**
|
|
78
|
+
* audit-timeout (2026-07-06) — per-tool result caps.
|
|
79
|
+
*
|
|
80
|
+
* Live false-FAIL #4: a c1.behavior audit failed with "both sap_inactive_objects
|
|
81
|
+
* results are truncated (only ZI_FRG3_DC and ZI_FRG_V2DC visible, followed by
|
|
82
|
+
* ellipsis)". The default 200-char cap beheaded the inactive-objects LIST,
|
|
83
|
+
* destroying the absence-from-inactive-list proof the D-F activeness arm depends
|
|
84
|
+
* on — and the auditor was CORRECTLY refusing to certify activeness from clipped
|
|
85
|
+
* evidence. The fix: render the inactive-objects list generously (it is names
|
|
86
|
+
* only, bounded in practice) so absence stays judgeable; only if it is
|
|
87
|
+
* pathologically huge do we clip it AND mark the clip HONESTLY, so the contract
|
|
88
|
+
* can fall back to the object_structure "version": "active" proof rather than
|
|
89
|
+
* inferring absence from a beheaded list. ATC results get a modest bump because
|
|
90
|
+
* the auditor reads finding priorities/counts from them. Everything else keeps
|
|
91
|
+
* the 200-char cap (object_structure surfaces "version" early, so 200 is fine).
|
|
92
|
+
*/
|
|
93
|
+
const INACTIVE_LIST_CAP = 4000;
|
|
94
|
+
const ATC_RESULT_CAP = 500;
|
|
95
|
+
/** Appended when the inactive-objects list itself is clipped — see the contract. */
|
|
96
|
+
const INACTIVE_CLIP_MARKER = ' … (list clipped — absence not verifiable from this row; use object_structure version)';
|
|
97
|
+
function resultCapFor(toolName) {
|
|
98
|
+
if (/^sap_inactive_objects/.test(toolName))
|
|
99
|
+
return INACTIVE_LIST_CAP;
|
|
100
|
+
if (/^sap_atc_run/.test(toolName))
|
|
101
|
+
return ATC_RESULT_CAP;
|
|
102
|
+
return RESULT_EXCERPT_CHARS;
|
|
103
|
+
}
|
|
104
|
+
function normalizeRow(raw) {
|
|
105
|
+
if (raw === null || typeof raw !== 'object')
|
|
106
|
+
return null;
|
|
107
|
+
const r = raw;
|
|
108
|
+
const toolName = typeof r.toolName === 'string' ? r.toolName
|
|
109
|
+
: typeof r.tool === 'string' ? r.tool
|
|
110
|
+
: null;
|
|
111
|
+
if (!toolName)
|
|
112
|
+
return null;
|
|
113
|
+
const isError = typeof r.isError === 'boolean' ? r.isError
|
|
114
|
+
: typeof r.ok === 'boolean' ? !r.ok
|
|
115
|
+
: false;
|
|
116
|
+
const at = typeof r.completedAt === 'string' ? r.completedAt
|
|
117
|
+
: typeof r.at === 'string' ? r.at
|
|
118
|
+
: null;
|
|
119
|
+
// PROTECTED is checked first: it is a strict-enough family that no
|
|
120
|
+
// existence-read name can collide with it, but ordering makes the tiering
|
|
121
|
+
// explicit for the reader.
|
|
122
|
+
const kind = PROTECTED_TOOL_RE.test(toolName) ? 'protected' : 'existence';
|
|
123
|
+
return { toolName, args: r.args, result: r.result, isError, at, kind };
|
|
124
|
+
}
|
|
125
|
+
/** Render the args as `name/type` when available, else a compact fallback. */
|
|
126
|
+
function renderArgs(args) {
|
|
127
|
+
if (args === null || args === undefined || typeof args !== 'object')
|
|
128
|
+
return '';
|
|
129
|
+
const a = args;
|
|
130
|
+
const name = a.name ?? a.object_name;
|
|
131
|
+
const type = a.type ?? a.object_type;
|
|
132
|
+
if (typeof name === 'string' && name.length > 0) {
|
|
133
|
+
return typeof type === 'string' && type.length > 0 ? `${name}/${type}` : name;
|
|
134
|
+
}
|
|
135
|
+
// No name/type — render what's available, compactly.
|
|
136
|
+
try {
|
|
137
|
+
const json = JSON.stringify(a);
|
|
138
|
+
if (!json || json === '{}')
|
|
139
|
+
return '';
|
|
140
|
+
return json.length <= ARGS_COMPACT_CHARS ? json : json.slice(0, ARGS_COMPACT_CHARS) + '…';
|
|
141
|
+
}
|
|
142
|
+
catch {
|
|
143
|
+
return '';
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
/**
|
|
147
|
+
* Result excerpt, flattened to a single line, capped per-tool (see
|
|
148
|
+
* resultCapFor). For sap_inactive_objects, a clip is marked honestly so a
|
|
149
|
+
* beheaded list can never masquerade as a complete absence proof.
|
|
150
|
+
*/
|
|
151
|
+
function renderResult(result, toolName) {
|
|
152
|
+
let text;
|
|
153
|
+
if (typeof result === 'string') {
|
|
154
|
+
text = result;
|
|
155
|
+
}
|
|
156
|
+
else if (result === null || result === undefined) {
|
|
157
|
+
text = '';
|
|
158
|
+
}
|
|
159
|
+
else {
|
|
160
|
+
try {
|
|
161
|
+
text = JSON.stringify(result) ?? '';
|
|
162
|
+
}
|
|
163
|
+
catch {
|
|
164
|
+
text = String(result);
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
const flat = text.replace(/\s+/g, ' ').trim();
|
|
168
|
+
const cap = resultCapFor(toolName);
|
|
169
|
+
if (flat.length <= cap)
|
|
170
|
+
return flat;
|
|
171
|
+
if (/^sap_inactive_objects/.test(toolName)) {
|
|
172
|
+
// Absence-from-inactive-list is an activeness PROOF; a bare ellipsis would
|
|
173
|
+
// silently behead the list and make absence unverifiable. Mark it honestly.
|
|
174
|
+
return flat.slice(0, cap) + INACTIVE_CLIP_MARKER;
|
|
175
|
+
}
|
|
176
|
+
return flat.slice(0, cap);
|
|
177
|
+
}
|
|
178
|
+
function renderRow(row) {
|
|
179
|
+
const status = row.isError ? 'ERROR' : 'ok';
|
|
180
|
+
return `${row.toolName}(${renderArgs(row.args)}) → ${status} — ${renderResult(row.result, row.toolName)}`;
|
|
181
|
+
}
|
|
182
|
+
/**
|
|
183
|
+
* Extract a bounded, auditor-ready excerpt of the session's write +
|
|
184
|
+
* verification tool calls from the JSONL audit log.
|
|
185
|
+
*
|
|
186
|
+
* - Only rows whose toolName matches {@link EVIDENCE_TOOL_RE} are included.
|
|
187
|
+
* - D-B: when `sinceIso` is given, rows whose timestamp is strictly BEFORE it
|
|
188
|
+
* are excluded — so an audit judging attempt N of a phase never sees the
|
|
189
|
+
* rows of an earlier attempt that shares the same session JSONL (two
|
|
190
|
+
* attempts in one CLI process). Untimestamped rows are kept (fail-open).
|
|
191
|
+
* - D-F two-tier cap: chronological order is preserved (newest-last). When
|
|
192
|
+
* the evidence exceeds `maxRows` (default 60), the OLDEST existence-read
|
|
193
|
+
* rows are dropped first; a PROTECTED (write / verify-proof) row is dropped
|
|
194
|
+
* only if no existence-read rows remain to drop. So a write row is never
|
|
195
|
+
* evicted to make room for a read — the exact regression a single bucket +
|
|
196
|
+
* the new existence reads would have introduced. The 60 cap is kept (not
|
|
197
|
+
* raised): the tiering, not a bigger budget, is what guarantees write
|
|
198
|
+
* survival, and a fixed cap keeps the audit prompt's token cost bounded.
|
|
199
|
+
* - Missing file/dir returns the sentinel — never throws.
|
|
200
|
+
* - Malformed JSONL lines are skipped silently.
|
|
201
|
+
*/
|
|
202
|
+
export async function extractAuditEvidence(sessionId, opts) {
|
|
203
|
+
const maxRows = opts?.maxRows ?? DEFAULT_MAX_ROWS;
|
|
204
|
+
const sinceIso = opts?.sinceIso;
|
|
205
|
+
const filePath = path.join(checkpointsRoot(), sessionId, 'tool-calls.jsonl');
|
|
206
|
+
let content;
|
|
207
|
+
try {
|
|
208
|
+
content = await fs.readFile(filePath, 'utf-8');
|
|
209
|
+
}
|
|
210
|
+
catch {
|
|
211
|
+
return NO_LOG_SENTINEL;
|
|
212
|
+
}
|
|
213
|
+
const rows = [];
|
|
214
|
+
for (const line of content.split('\n')) {
|
|
215
|
+
const trimmed = line.trim();
|
|
216
|
+
if (!trimmed)
|
|
217
|
+
continue;
|
|
218
|
+
let parsed;
|
|
219
|
+
try {
|
|
220
|
+
parsed = JSON.parse(trimmed);
|
|
221
|
+
}
|
|
222
|
+
catch {
|
|
223
|
+
continue; // malformed line — skip silently
|
|
224
|
+
}
|
|
225
|
+
const row = normalizeRow(parsed);
|
|
226
|
+
if (!row || !EVIDENCE_TOOL_RE.test(row.toolName))
|
|
227
|
+
continue;
|
|
228
|
+
// D-B attempt-window: drop rows recorded before this attempt started.
|
|
229
|
+
// ISO-8601 UTC strings compare correctly lexicographically. A row with no
|
|
230
|
+
// timestamp is never excluded (fail-open).
|
|
231
|
+
if (sinceIso && row.at !== null && row.at < sinceIso)
|
|
232
|
+
continue;
|
|
233
|
+
rows.push(row);
|
|
234
|
+
}
|
|
235
|
+
if (rows.length === 0)
|
|
236
|
+
return NO_EVIDENCE_ROWS;
|
|
237
|
+
return capRows(rows, maxRows).map(renderRow).join('\n');
|
|
238
|
+
}
|
|
239
|
+
/**
|
|
240
|
+
* D-F two-tier cap. Returns at most `maxRows` rows in chronological order.
|
|
241
|
+
* Eviction order when over budget: oldest EXISTENCE reads first, then (only
|
|
242
|
+
* if still over and no reads remain) oldest PROTECTED rows. Writes are thus
|
|
243
|
+
* never evicted by reads.
|
|
244
|
+
*/
|
|
245
|
+
function capRows(rows, maxRows) {
|
|
246
|
+
if (rows.length <= maxRows)
|
|
247
|
+
return rows;
|
|
248
|
+
let over = rows.length - maxRows;
|
|
249
|
+
const dropped = new Set();
|
|
250
|
+
// Pass 1 — drop oldest existence reads.
|
|
251
|
+
for (let i = 0; i < rows.length && over > 0; i++) {
|
|
252
|
+
if (rows[i].kind === 'existence') {
|
|
253
|
+
dropped.add(i);
|
|
254
|
+
over--;
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
// Pass 2 — reads exhausted, still over: drop oldest protected rows (writes
|
|
258
|
+
// evicting older writes is the original 60-cap behavior, not a read winning).
|
|
259
|
+
for (let i = 0; i < rows.length && over > 0; i++) {
|
|
260
|
+
if (!dropped.has(i)) {
|
|
261
|
+
dropped.add(i);
|
|
262
|
+
over--;
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
return rows.filter((_, i) => !dropped.has(i));
|
|
266
|
+
}
|