@awak-app/simy-cli 0.1.2 → 0.1.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,7 +4,9 @@ import {
4
4
  buildIndependentAuditInstruction,
5
5
  } from "./independent-audit.js";
6
6
  import { buildCodingInstruction, buildPromptInterventions } from "./instruction.js";
7
+ import { canStartProvider, refreshBudgetState } from "./budget.js";
7
8
  import { buildAttempt, mergeTokenUsage } from "./result.js";
9
+ import { evaluateRetryCircuit } from "./retry.js";
8
10
  import { appendEvent, publish } from "./shared.js";
9
11
 
10
12
  export async function runCodingLoop({
@@ -54,8 +56,22 @@ export async function runCodingLoop({
54
56
  attemptNumber += 1
55
57
  ) {
56
58
  if (shouldStop()) return stopCodingLoop(snapshot, onUpdate);
59
+ const executorBudget = canStartProvider(snapshot, {
60
+ attemptNumber,
61
+ phase: "executor",
62
+ });
63
+ if (!executorBudget.allowed) {
64
+ await stopForBudget({
65
+ snapshot,
66
+ attempt: snapshot.attempts.at(-1) ?? null,
67
+ reason: executorBudget.reason,
68
+ onUpdate,
69
+ });
70
+ return snapshot;
71
+ }
57
72
  const queuedGuidance = consumeHumanGuidance();
58
73
  const attemptGuidance = [humanGuidance, queuedGuidance].filter(Boolean).join("\n\n");
74
+ const previousAttempt = snapshot.attempts.at(-1) ?? null;
59
75
  const promptInterventions = buildPromptInterventions(snapshot.charter, {
60
76
  attemptNumber,
61
77
  previousFindings,
@@ -63,7 +79,8 @@ export async function runCodingLoop({
63
79
  });
64
80
  const instruction = buildCodingInstruction(snapshot.charter, {
65
81
  attemptNumber,
66
- previousAttempt: snapshot.attempts.at(-1) ?? null,
82
+ budget: executorBudget.budget,
83
+ previousAttempt,
67
84
  previousFindings,
68
85
  promptInterventions,
69
86
  });
@@ -98,6 +115,7 @@ export async function runCodingLoop({
98
115
  promptInterventions,
99
116
  execution,
100
117
  });
118
+ attempt.budget = refreshBudgetState(snapshot, attempt);
101
119
 
102
120
  appendEvent(snapshot, "collecting_evidence", `Collecting attempt ${attemptNumber} evidence.`, {
103
121
  attempt_number: attemptNumber,
@@ -109,6 +127,7 @@ export async function runCodingLoop({
109
127
  });
110
128
  await publish(snapshot, "collecting_evidence", onUpdate);
111
129
  attempt.observed_evidence = await safeEvidence(collectEvidence, attempt);
130
+ if (shouldStop()) return stopCodingLoop(snapshot, onUpdate);
112
131
 
113
132
  appendEvent(snapshot, "auditing", `Verifying local evidence for attempt ${attemptNumber}.`, {
114
133
  attempt_number: attemptNumber,
@@ -121,9 +140,21 @@ export async function runCodingLoop({
121
140
  attempt.observed_evidence?.github?.available === true,
122
141
  });
123
142
  await publish(snapshot, "auditing", onUpdate);
124
- attempt.audit = await auditAttempt(snapshot.charter, attempt);
143
+ attempt.audit = await auditAttempt(snapshot.charter, attempt, { previousAttempt });
144
+ if (shouldStop()) return stopCodingLoop(snapshot, onUpdate);
125
145
  attempt.implementation_gate = attempt.audit.implementation_gate;
126
146
  snapshot.attempts.push(attempt);
147
+ attempt.budget = refreshBudgetState(snapshot);
148
+
149
+ if (snapshot.budget.token_exhausted) {
150
+ await stopForBudget({
151
+ snapshot,
152
+ attempt,
153
+ reason: "token_budget_exhausted",
154
+ onUpdate,
155
+ });
156
+ return snapshot;
157
+ }
127
158
 
128
159
  if (!attempt.audit.passed) {
129
160
  snapshot.final_audit = attempt.audit;
@@ -140,9 +171,23 @@ export async function runCodingLoop({
140
171
  continue;
141
172
  }
142
173
 
174
+ const reviewerBudget = canStartProvider(snapshot, {
175
+ attemptNumber,
176
+ phase: "reviewer",
177
+ });
178
+ if (!reviewerBudget.allowed) {
179
+ await stopForBudget({
180
+ snapshot,
181
+ attempt,
182
+ reason: reviewerBudget.reason,
183
+ onUpdate,
184
+ });
185
+ return snapshot;
186
+ }
143
187
  appendEvent(snapshot, "independent_auditing", "Starting an independent AI audit.", {
144
188
  attempt_number: attemptNumber,
145
189
  backend: snapshot.charter.audit_backend,
190
+ budget: reviewerBudget.budget,
146
191
  });
147
192
  await publish(snapshot, "independent_auditing", onUpdate);
148
193
  if (shouldStop()) return stopCodingLoop(snapshot, onUpdate);
@@ -159,10 +204,13 @@ export async function runCodingLoop({
159
204
  if (shouldStop()) return stopCodingLoop(snapshot, onUpdate);
160
205
  attempt.independent_audit = buildIndependentAudit(auditExecution);
161
206
  attempt.token_usage = mergeTokenUsage(attempt.token_usage, auditExecution.tokenUsage);
207
+ attempt.budget = refreshBudgetState(snapshot);
162
208
 
163
209
  // Re-collect after the read-only auditor to catch any mutated HEAD or working tree.
164
210
  attempt.observed_evidence = await safeEvidence(collectEvidence, attempt);
165
- attempt.audit = await auditAttempt(snapshot.charter, attempt);
211
+ if (shouldStop()) return stopCodingLoop(snapshot, onUpdate);
212
+ attempt.audit = await auditAttempt(snapshot.charter, attempt, { previousAttempt });
213
+ if (shouldStop()) return stopCodingLoop(snapshot, onUpdate);
166
214
  attempt.implementation_gate = attempt.audit.implementation_gate;
167
215
 
168
216
  const independentFindings = attempt.independent_audit.findings || [];
@@ -192,10 +240,12 @@ export async function runCodingLoop({
192
240
  attempt.independent_audit.findings || [],
193
241
  );
194
242
  snapshot.done_gate_passed = attempt.pr_readiness.ready;
243
+ attempt.budget = refreshBudgetState(snapshot);
195
244
  appendEvent(snapshot, attempt.pr_readiness.state, attempt.pr_readiness.summary, {
196
245
  attempt_number: attemptNumber,
197
246
  pr_url: attempt.pr_url,
198
247
  pending_reasons: attempt.pr_readiness.pending_reasons,
248
+ budget: snapshot.budget,
199
249
  });
200
250
  await publish(snapshot, attempt.pr_readiness.state, onUpdate);
201
251
  return snapshot;
@@ -206,7 +256,7 @@ export async function runCodingLoop({
206
256
 
207
257
  async function stopCodingLoop(snapshot, onUpdate) {
208
258
  if (snapshot.state === "stopped") return snapshot;
209
- appendEvent(snapshot, "stopped", "Coding loop stopped by the local human operator.", {
259
+ appendEvent(snapshot, "stopped", "Agentic loop stopped by the local human operator.", {
210
260
  source: "local_cli",
211
261
  });
212
262
  await publish(snapshot, "stopped", onUpdate);
@@ -221,7 +271,9 @@ export async function recheckPrReadiness({ snapshot, collectEvidence, onUpdate }
221
271
  });
222
272
  await publish(snapshot, "checking_pr", onUpdate);
223
273
  attempt.observed_evidence = await safeEvidence(collectEvidence, attempt);
224
- attempt.audit = await auditAttempt(snapshot.charter, attempt);
274
+ attempt.audit = await auditAttempt(snapshot.charter, attempt, {
275
+ previousAttempt: snapshot.attempts.at(-2) ?? null,
276
+ });
225
277
  snapshot.implementation_gate_passed = attempt.audit.passed;
226
278
  attempt.pr_readiness = evaluatePrReadiness(snapshot.charter, attempt);
227
279
  attempt.pr_gate = attempt.pr_readiness;
@@ -326,6 +378,60 @@ async function handleFailedAudit({
326
378
  onUpdate,
327
379
  }) {
328
380
  snapshot.charter.thread_state.failure_count += 1;
381
+ const budget = refreshBudgetState(snapshot);
382
+ if (budget.token_exhausted) {
383
+ return stopForBudget({
384
+ snapshot,
385
+ attempt,
386
+ reason: "token_budget_exhausted",
387
+ onUpdate,
388
+ });
389
+ }
390
+ const retryCircuit = evaluateRetryCircuit({
391
+ attempt,
392
+ previousAttempt: snapshot.attempts.at(-2) ?? null,
393
+ findings,
394
+ });
395
+ if (retryCircuit.open) {
396
+ const avoidedAttempts = Math.max(snapshot.charter.max_attempts - attemptNumber, 0);
397
+ const circuitState = {
398
+ state: "open",
399
+ reason: "identical_commit_and_findings",
400
+ opened_at: new Date().toISOString(),
401
+ attempt_number: attemptNumber,
402
+ previous_attempt_number: retryCircuit.previous_attempt_number,
403
+ avoided_attempts: avoidedAttempts,
404
+ ...retryCircuit.fingerprint,
405
+ };
406
+ snapshot.charter.thread_state.retry_circuit_breaker = circuitState;
407
+ snapshot.final_audit = retryCircuitAudit(snapshot.final_audit, circuitState);
408
+ appendEvent(
409
+ snapshot,
410
+ "blocked",
411
+ "Retry stopped because the commit and findings did not change.",
412
+ {
413
+ code: "retry_circuit_open",
414
+ reason: circuitState.reason,
415
+ attempt_number: attemptNumber,
416
+ previous_attempt_number: retryCircuit.previous_attempt_number,
417
+ commit_sha: circuitState.commit_sha,
418
+ finding_codes: circuitState.finding_codes,
419
+ retry_fingerprint: circuitState.digest,
420
+ avoided_attempts: avoidedAttempts,
421
+ },
422
+ );
423
+ await publish(snapshot, "blocked", onUpdate);
424
+ return true;
425
+ }
426
+ if (attemptNumber >= snapshot.charter.max_attempts) {
427
+ await stopForBudget({
428
+ snapshot,
429
+ attempt,
430
+ reason: "retry_budget_exhausted",
431
+ onUpdate,
432
+ });
433
+ return true;
434
+ }
329
435
  if (requiresHuman) {
330
436
  appendEvent(snapshot, "waiting_human", "PR audit requires human input.", {
331
437
  finding_codes: findings.map((finding) => finding.code),
@@ -341,13 +447,85 @@ async function handleFailedAudit({
341
447
  await publish(snapshot, "re_instructing", onUpdate);
342
448
  return false;
343
449
  }
344
- appendEvent(snapshot, "blocked", "Coding loop exhausted its attempt budget.", {
345
- finding_codes: findings.map((finding) => finding.code),
450
+ return false;
451
+ }
452
+
453
+ async function stopForBudget({ snapshot, attempt, reason, onUpdate }) {
454
+ const budget = refreshBudgetState(snapshot);
455
+ const finding = budgetFinding(reason, budget);
456
+ const existingAudit = snapshot.final_audit || attempt?.audit || {};
457
+ const findings = Array.isArray(existingAudit.findings) ? existingAudit.findings : [];
458
+ snapshot.final_audit = {
459
+ ...existingAudit,
460
+ passed: false,
461
+ summary: finding.explanation,
462
+ findings: findings.some((item) => item.code === finding.code)
463
+ ? findings
464
+ : [...findings, finding],
465
+ auto_repairable: false,
466
+ requires_human: true,
467
+ budget,
468
+ };
469
+ if (attempt) attempt.budget = budget;
470
+ const message =
471
+ reason === "token_budget_exhausted"
472
+ ? "Agentic loop exhausted its token budget."
473
+ : "Agentic loop exhausted its retry budget.";
474
+ appendEvent(snapshot, "blocked", message, {
475
+ code: reason,
476
+ attempt_number: attempt?.attempt_number ?? null,
477
+ finding_codes: [finding.code],
478
+ budget,
346
479
  });
347
480
  await publish(snapshot, "blocked", onUpdate);
348
481
  return true;
349
482
  }
350
483
 
484
+ function budgetFinding(reason, budget) {
485
+ if (reason === "token_budget_exhausted") {
486
+ return {
487
+ passed: false,
488
+ code: "TOKEN_BUDGET_EXHAUSTED",
489
+ severity: "blocker",
490
+ target: "token_budget",
491
+ explanation: `The run used ${budget.tokens_used} of ${budget.token_budget} provider tokens, so no additional executor or reviewer call can start.`,
492
+ repairability: "manual",
493
+ auto_fix_hint: null,
494
+ };
495
+ }
496
+ return {
497
+ passed: false,
498
+ code: "RETRY_BUDGET_EXHAUSTED",
499
+ severity: "blocker",
500
+ target: "retry_budget",
501
+ explanation: `The run used ${budget.retries_used} of ${budget.retry_budget} automatic retries.`,
502
+ repairability: "manual",
503
+ auto_fix_hint: null,
504
+ };
505
+ }
506
+
507
+ function retryCircuitAudit(audit, circuitState) {
508
+ const circuitFinding = {
509
+ passed: false,
510
+ code: "RETRY_CIRCUIT_OPEN",
511
+ severity: "blocker",
512
+ target: "retry_loop",
513
+ explanation:
514
+ "The last two attempts produced the same commit and findings, so another automatic retry would repeat the same work.",
515
+ repairability: "manual",
516
+ auto_fix_hint: null,
517
+ };
518
+ return {
519
+ ...(audit || {}),
520
+ passed: false,
521
+ summary: "Automatic retries stopped because the last two failure results were identical.",
522
+ findings: [...(Array.isArray(audit?.findings) ? audit.findings : []), circuitFinding],
523
+ auto_repairable: false,
524
+ requires_human: true,
525
+ retry_circuit_breaker: circuitState,
526
+ };
527
+ }
528
+
351
529
  function aggregateAudit(attempt, passed, findings) {
352
530
  return {
353
531
  passed,
@@ -106,7 +106,7 @@ export function summarizeCodingLoopEvent(snapshot, event) {
106
106
  };
107
107
  case "blocked":
108
108
  return {
109
- summary: "The run cannot continue automatically with its current attempt budget or evidence.",
109
+ summary: "The run cannot continue automatically with its current token, retry, or evidence budget.",
110
110
  result: findings.length ? findingResult(findings) : cleanString(event?.message),
111
111
  next: "Review the findings, adjust the request or evidence, and resume the run.",
112
112
  };
@@ -118,7 +118,7 @@ export function summarizeCodingLoopEvent(snapshot, event) {
118
118
  };
119
119
  case "stopped":
120
120
  return {
121
- summary: "The local human operator stopped this coding run.",
121
+ summary: "The local human operator stopped this Agentic Loop run.",
122
122
  result: "No further executor handoff will occur for this run.",
123
123
  next: "Start a new run or explicitly continue this one when ready.",
124
124
  };
@@ -0,0 +1,129 @@
1
+ import { cleanString, stringArray } from "./shared.js";
2
+
3
+ const MIN_RETRY_HYPOTHESES = 8;
4
+
5
+ export function normalizeProblemSolving(value) {
6
+ const record = value && typeof value === "object" && !Array.isArray(value) ? value : {};
7
+ const hypotheses = Array.isArray(record.hypotheses)
8
+ ? record.hypotheses.map(normalizeHypothesis).filter(Boolean)
9
+ : [];
10
+ return {
11
+ hypotheses,
12
+ selected_hypothesis:
13
+ cleanString(record.selected_hypothesis_id) ||
14
+ cleanString(record.selected_hypothesis) ||
15
+ null,
16
+ evidence_checked: uniqueStrings(record.evidence_checked),
17
+ strategy: cleanString(record.strategy) || null,
18
+ changed_from_previous: cleanString(record.changed_from_previous) || null,
19
+ };
20
+ }
21
+
22
+ export function evaluateRetryProblemSolving(attempt, previousAttempt) {
23
+ const current = normalizeProblemSolving(
24
+ attempt?.problem_solving || attempt?.raw_output?.problem_solving,
25
+ );
26
+ const previous = normalizeProblemSolving(
27
+ previousAttempt?.problem_solving || previousAttempt?.raw_output?.problem_solving,
28
+ );
29
+ if (!Number.isInteger(attempt?.attempt_number) || attempt.attempt_number <= 1) {
30
+ return { required: false, passed: true, checks: [], current, previous };
31
+ }
32
+
33
+ const distinctHypotheses = new Set(
34
+ current.hypotheses.map((hypothesis) => canonical(hypothesis.statement)),
35
+ ).size;
36
+ const selectedMatches = current.hypotheses.some(
37
+ (hypothesis) =>
38
+ canonical(hypothesis.id) === canonical(current.selected_hypothesis) ||
39
+ canonical(hypothesis.statement) === canonical(current.selected_hypothesis),
40
+ );
41
+ const changeDescriptionIsMeaningful = Boolean(
42
+ current.changed_from_previous &&
43
+ !/^(?:same|unchanged|none|n\/?a|no change)$/i.test(current.changed_from_previous),
44
+ );
45
+ const strategyDiffers = Boolean(
46
+ current.strategy &&
47
+ (!previous.strategy || canonical(current.strategy) !== canonical(previous.strategy)),
48
+ );
49
+ const checks = [
50
+ {
51
+ id: "retry_hypotheses_expanded",
52
+ label: "Retry considers at least eight distinct hypotheses",
53
+ passed: distinctHypotheses >= MIN_RETRY_HYPOTHESES,
54
+ detail: `Expected at least ${MIN_RETRY_HYPOTHESES} distinct hypotheses; recorded ${distinctHypotheses}.`,
55
+ },
56
+ {
57
+ id: "retry_hypothesis_selected",
58
+ label: "Retry selects one recorded hypothesis",
59
+ passed: Boolean(current.selected_hypothesis && selectedMatches),
60
+ detail: current.selected_hypothesis
61
+ ? `Selected hypothesis ${current.selected_hypothesis} must match a recorded hypothesis.`
62
+ : "No selected hypothesis was recorded.",
63
+ },
64
+ {
65
+ id: "retry_evidence_checked",
66
+ label: "Retry records discriminating evidence",
67
+ passed: current.evidence_checked.length > 0,
68
+ detail:
69
+ current.evidence_checked.length > 0
70
+ ? `${current.evidence_checked.length} evidence item(s) recorded.`
71
+ : "No evidence checked against the hypotheses was recorded.",
72
+ },
73
+ {
74
+ id: "retry_strategy_changed",
75
+ label: "Retry uses a changed strategy",
76
+ passed: Boolean(current.strategy && changeDescriptionIsMeaningful && strategyDiffers),
77
+ detail: retryStrategyDetail({ current, previous, strategyDiffers }),
78
+ },
79
+ ];
80
+ return {
81
+ required: true,
82
+ passed: checks.every((check) => check.passed),
83
+ checks,
84
+ current,
85
+ previous,
86
+ };
87
+ }
88
+
89
+ export function problemSolvingResultContract() {
90
+ return [
91
+ "problem_solving must be an object with hypotheses, selected_hypothesis_id, evidence_checked, strategy, and changed_from_previous.",
92
+ "Each hypothesis must use {id, statement}; evidence_checked must list concrete observations or commands.",
93
+ "On attempt 1, record the assumptions, hypotheses considered, selected hypothesis, and implementation strategy; changed_from_previous may be null.",
94
+ `On every retry, record at least ${MIN_RETRY_HYPOTHESES} distinct MECE hypotheses, select one of them, check discriminating evidence, use a strategy different from the previous attempt, and explain the change in changed_from_previous.`,
95
+ ].join("\n");
96
+ }
97
+
98
+ function normalizeHypothesis(value, index) {
99
+ if (typeof value === "string") {
100
+ const statement = cleanString(value);
101
+ return statement ? { id: `H${index + 1}`, statement } : null;
102
+ }
103
+ if (!value || typeof value !== "object" || Array.isArray(value)) return null;
104
+ const statement = cleanString(value.statement) || cleanString(value.hypothesis);
105
+ if (!statement) return null;
106
+ return {
107
+ id: cleanString(value.id) || `H${index + 1}`,
108
+ statement,
109
+ };
110
+ }
111
+
112
+ function uniqueStrings(value) {
113
+ return [...new Set(stringArray(value))];
114
+ }
115
+
116
+ function canonical(value) {
117
+ return cleanString(value).toLowerCase().replace(/\s+/g, " ");
118
+ }
119
+
120
+ function retryStrategyDetail({ current, previous, strategyDiffers }) {
121
+ if (!current.strategy) return "No retry strategy was recorded.";
122
+ if (!current.changed_from_previous) {
123
+ return "The retry did not explain what changed from the previous attempt.";
124
+ }
125
+ if (!strategyDiffers) {
126
+ return `The retry repeated the previous strategy: ${current.strategy}`;
127
+ }
128
+ return `Changed from ${previous.strategy || "the prior unstructured approach"} to ${current.strategy}: ${current.changed_from_previous}`;
129
+ }