@mjasnikovs/pi-task 0.38.29 → 0.38.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config/config.d.ts +70 -70
- package/dist/config/config.js +26 -35
- package/dist/config/extension-list.d.ts +6 -5
- package/dist/config/extension-list.js +3 -2
- package/dist/config/reasoning-args.d.ts +9 -7
- package/dist/config/reasoning-args.js +12 -10
- package/dist/config/reasoning.d.ts +44 -105
- package/dist/config/reasoning.js +27 -704
- package/dist/config/register.d.ts +34 -48
- package/dist/config/register.js +41 -51
- package/dist/config/tool-list.d.ts +16 -16
- package/dist/config/tool-list.js +1 -1
- package/dist/remote/bridge.d.ts +19 -10
- package/dist/remote/bridge.js +3 -2
- package/dist/remote/broadcast.js +3 -1
- package/dist/remote/events.js +12 -11
- package/dist/remote/history.d.ts +1 -1
- package/dist/remote/protocol.d.ts +6 -3
- package/dist/remote/protocol.js +2 -1
- package/dist/remote/push.d.ts +16 -16
- package/dist/remote/push.js +27 -27
- package/dist/remote/register.d.ts +3 -3
- package/dist/remote/register.js +17 -19
- package/dist/remote/server.d.ts +9 -8
- package/dist/remote/server.js +15 -14
- package/dist/remote/session-state.d.ts +5 -4
- package/dist/remote/session-state.js +8 -5
- package/dist/remote/sw.d.ts +7 -6
- package/dist/remote/sw.js +7 -6
- package/dist/remote/tailscale.d.ts +4 -2
- package/dist/remote/tailscale.js +4 -2
- package/dist/remote/ui-highlight.js +6 -5
- package/dist/remote/ui-render.js +4 -4
- package/dist/remote/ui-script.js +24 -24
- package/dist/remote/ui-styles.d.ts +1 -1
- package/dist/remote/ui-styles.js +10 -13
- package/dist/remote/ui-tools.js +9 -6
- package/dist/shared/child-extensions.d.ts +29 -17
- package/dist/shared/child-extensions.js +29 -17
- package/dist/shared/child-output.d.ts +30 -24
- package/dist/shared/child-output.js +25 -17
- package/dist/shared/child-process.d.ts +47 -40
- package/dist/shared/child-process.js +50 -59
- package/dist/shared/command-watchdog.d.ts +22 -16
- package/dist/shared/command-watchdog.js +28 -21
- package/dist/shared/fs-text.d.ts +16 -10
- package/dist/shared/fs-text.js +16 -10
- package/dist/shared/git-runner.d.ts +25 -25
- package/dist/shared/git-runner.js +25 -25
- package/dist/shared/leaked-tool-call.d.ts +17 -11
- package/dist/shared/leaked-tool-call.js +23 -15
- package/dist/shared/model-endpoint.d.ts +29 -16
- package/dist/shared/model-endpoint.js +33 -21
- package/dist/shared/pi-invocation.d.ts +7 -4
- package/dist/shared/pi-invocation.js +12 -7
- package/dist/shared/pkg-version.d.ts +13 -5
- package/dist/shared/pkg-version.js +13 -5
- package/dist/shared/reasoning-capability.d.ts +35 -24
- package/dist/shared/reasoning-capability.js +35 -24
- package/dist/shared/stream-watchdog.d.ts +60 -44
- package/dist/shared/stream-watchdog.js +62 -45
- package/dist/task/accept-debt.d.ts +41 -43
- package/dist/task/accept-debt.js +73 -65
- package/dist/task/api-synthesis.d.ts +24 -21
- package/dist/task/api-synthesis.js +32 -26
- package/dist/task/apis-contract.d.ts +32 -64
- package/dist/task/apis-contract.js +32 -64
- package/dist/task/artifact-closure.d.ts +27 -13
- package/dist/task/artifact-closure.js +95 -67
- package/dist/task/auto-commit.d.ts +46 -35
- package/dist/task/auto-commit.js +51 -38
- package/dist/task/auto-io.d.ts +45 -25
- package/dist/task/auto-io.js +57 -29
- package/dist/task/auto-orchestrator.d.ts +26 -24
- package/dist/task/auto-orchestrator.js +178 -162
- package/dist/task/auto-prompts.d.ts +36 -24
- package/dist/task/auto-prompts.js +40 -26
- package/dist/task/autofix-ledger.d.ts +27 -25
- package/dist/task/autofix-ledger.js +29 -26
- package/dist/task/batch-test-task.d.ts +20 -12
- package/dist/task/batch-test-task.js +67 -60
- package/dist/task/boot-probe.d.ts +60 -44
- package/dist/task/boot-probe.js +91 -72
- package/dist/task/cancel-input.d.ts +30 -16
- package/dist/task/cancel-input.js +20 -11
- package/dist/task/cancel-points.d.ts +27 -20
- package/dist/task/cancel-points.js +30 -22
- package/dist/task/child-runner.d.ts +46 -51
- package/dist/task/child-runner.js +48 -49
- package/dist/task/child-status.d.ts +23 -16
- package/dist/task/child-status.js +23 -16
- package/dist/task/clamp-output.js +12 -5
- package/dist/task/command-run.d.ts +31 -28
- package/dist/task/command-run.js +44 -35
- package/dist/task/command-shrink.d.ts +25 -18
- package/dist/task/command-shrink.js +37 -31
- package/dist/task/command-watchdog.d.ts +9 -6
- package/dist/task/command-watchdog.js +21 -15
- package/dist/task/context-attribution.d.ts +34 -26
- package/dist/task/context-attribution.js +34 -26
- package/dist/task/context-silence.d.ts +39 -29
- package/dist/task/context-silence.js +35 -25
- package/dist/task/context-usage.d.ts +16 -9
- package/dist/task/context-usage.js +16 -9
- package/dist/task/contracts.d.ts +8 -4
- package/dist/task/contracts.js +25 -17
- package/dist/task/coverage-loop.d.ts +22 -18
- package/dist/task/coverage-loop.js +35 -30
- package/dist/task/critique-probes.d.ts +13 -14
- package/dist/task/critique-probes.js +50 -39
- package/dist/task/debug-log.d.ts +13 -5
- package/dist/task/debug-log.js +32 -20
- package/dist/task/decompose-fidelity.d.ts +11 -9
- package/dist/task/decompose-fidelity.js +38 -33
- package/dist/task/decompose-granularity.d.ts +41 -38
- package/dist/task/decompose-granularity.js +41 -38
- package/dist/task/deep-render-check.d.ts +22 -14
- package/dist/task/deep-render-check.js +40 -31
- package/dist/task/dropped-input.d.ts +12 -7
- package/dist/task/dropped-input.js +5 -2
- package/dist/task/enforce-attribution.d.ts +38 -47
- package/dist/task/enforce-attribution.js +46 -52
- package/dist/task/enforce-guidelines.d.ts +31 -20
- package/dist/task/enforce-guidelines.js +32 -21
- package/dist/task/enrichment.d.ts +7 -2
- package/dist/task/enrichment.js +26 -14
- package/dist/task/env-notes.d.ts +16 -7
- package/dist/task/env-notes.js +48 -31
- package/dist/task/env-template-closure.d.ts +4 -4
- package/dist/task/env-template-closure.js +42 -34
- package/dist/task/external-context.d.ts +28 -21
- package/dist/task/external-context.js +17 -12
- package/dist/task/failure-classifier.d.ts +4 -5
- package/dist/task/failure-classifier.js +6 -7
- package/dist/task/file-inventory.d.ts +15 -11
- package/dist/task/file-inventory.js +25 -22
- package/dist/task/final-gate-fix.d.ts +74 -86
- package/dist/task/final-gate-fix.js +97 -116
- package/dist/task/final-gate-progress.d.ts +29 -46
- package/dist/task/final-gate-progress.js +40 -51
- package/dist/task/final-gate.d.ts +64 -97
- package/dist/task/final-gate.js +192 -199
- package/dist/task/fix-child.d.ts +21 -27
- package/dist/task/fix-child.js +21 -27
- package/dist/task/foreign-path.d.ts +6 -5
- package/dist/task/foreign-path.js +0 -0
- package/dist/task/frozen-conflict.d.ts +9 -10
- package/dist/task/frozen-conflict.js +61 -64
- package/dist/task/frozen-path-guard.d.ts +35 -14
- package/dist/task/frozen-path-guard.js +56 -39
- package/dist/task/gate-child.d.ts +27 -28
- package/dist/task/gate-child.js +36 -35
- package/dist/task/gate-deps.d.ts +34 -27
- package/dist/task/gate-deps.js +169 -159
- package/dist/task/gate-tally.d.ts +77 -80
- package/dist/task/gate-tally.js +65 -68
- package/dist/task/git-state-guard.d.ts +15 -11
- package/dist/task/git-state-guard.js +76 -66
- package/dist/task/impl-widget.d.ts +25 -16
- package/dist/task/impl-widget.js +27 -17
- package/dist/task/implementation-thinking.d.ts +33 -31
- package/dist/task/implementation-thinking.js +5 -6
- package/dist/task/implementation-turn.d.ts +34 -31
- package/dist/task/implementation-turn.js +29 -27
- package/dist/task/inline-markdown.d.ts +20 -7
- package/dist/task/inline-markdown.js +15 -6
- package/dist/task/launch-config-gap.js +25 -39
- package/dist/task/launch-contract.d.ts +18 -21
- package/dist/task/launch-contract.js +28 -30
- package/dist/task/launch-manifest.d.ts +6 -2
- package/dist/task/launch-manifest.js +35 -34
- package/dist/task/ledger.js +16 -14
- package/dist/task/lint-fix.d.ts +6 -8
- package/dist/task/lint-fix.js +67 -69
- package/dist/task/loop-detector.d.ts +9 -8
- package/dist/task/loop-detector.js +16 -12
- package/dist/task/mid-run-input.d.ts +17 -15
- package/dist/task/mid-run-input.js +17 -15
- package/dist/task/orchestrator.d.ts +24 -28
- package/dist/task/orchestrator.js +62 -64
- package/dist/task/orientation.d.ts +18 -23
- package/dist/task/orientation.js +24 -31
- package/dist/task/owned-freeze-conflict.d.ts +21 -20
- package/dist/task/owned-freeze-conflict.js +52 -85
- package/dist/task/owned-freeze-reassign.d.ts +40 -60
- package/dist/task/owned-freeze-reassign.js +41 -61
- package/dist/task/parsers.d.ts +4 -2
- package/dist/task/parsers.js +4 -4
- package/dist/task/phases.d.ts +41 -48
- package/dist/task/phases.js +179 -248
- package/dist/task/plan-io.d.ts +6 -7
- package/dist/task/plan-io.js +6 -7
- package/dist/task/plan-orchestrator.d.ts +10 -8
- package/dist/task/plan-orchestrator.js +14 -10
- package/dist/task/plan-prompts.d.ts +6 -5
- package/dist/task/plan-prompts.js +6 -5
- package/dist/task/plan-readonly.d.ts +4 -5
- package/dist/task/plan-readonly.js +4 -5
- package/dist/task/plan-rounds.d.ts +17 -29
- package/dist/task/plan-rounds.js +21 -34
- package/dist/task/plan-session.d.ts +58 -72
- package/dist/task/plan-session.js +61 -83
- package/dist/task/probe-gaming.d.ts +28 -27
- package/dist/task/probe-gaming.js +0 -0
- package/dist/task/prohibition-probe.d.ts +14 -16
- package/dist/task/prompts.d.ts +3 -4
- package/dist/task/prompts.js +17 -26
- package/dist/task/qa-transcript.d.ts +15 -22
- package/dist/task/qa-transcript.js +15 -21
- package/dist/task/question-box.d.ts +17 -13
- package/dist/task/question-box.js +19 -15
- package/dist/task/question-dedup.d.ts +6 -7
- package/dist/task/question-dedup.js +13 -14
- package/dist/task/question-dialog.d.ts +22 -32
- package/dist/task/question-dialog.js +22 -32
- package/dist/task/question-source.d.ts +18 -44
- package/dist/task/question-source.js +22 -51
- package/dist/task/refuted-constraint.d.ts +11 -31
- package/dist/task/refuted-constraint.js +27 -51
- package/dist/task/regenerable-artifacts.d.ts +12 -31
- package/dist/task/regenerable-artifacts.js +12 -31
- package/dist/task/render-check.d.ts +11 -22
- package/dist/task/render-check.js +33 -46
- package/dist/task/repo-health-check.d.ts +10 -14
- package/dist/task/repo-health-check.js +17 -23
- package/dist/task/requirements.d.ts +38 -71
- package/dist/task/requirements.js +78 -126
- package/dist/task/research-fanout-budget.d.ts +51 -88
- package/dist/task/research-fanout-budget.js +51 -88
- package/dist/task/research-worker.d.ts +29 -39
- package/dist/task/research-worker.js +37 -61
- package/dist/task/resume-gap.d.ts +14 -15
- package/dist/task/root-cause-repair.d.ts +9 -9
- package/dist/task/root-cause-repair.js +28 -40
- package/dist/task/run-bracket.d.ts +10 -13
- package/dist/task/run-end.d.ts +12 -22
- package/dist/task/run-end.js +8 -16
- package/dist/task/run-final-gate.d.ts +19 -21
- package/dist/task/run-final-gate.js +62 -80
- package/dist/task/runner-globs.d.ts +12 -13
- package/dist/task/runner-globs.js +12 -13
- package/dist/task/runner-resolve.d.ts +9 -9
- package/dist/task/runner-resolve.js +22 -23
- package/dist/task/script-escape.d.ts +10 -12
- package/dist/task/script-escape.js +13 -14
- package/dist/task/serve-entry.d.ts +1 -1
- package/dist/task/serve-entry.js +22 -25
- package/dist/task/service-blocks.js +4 -2
- package/dist/task/shipped-source.d.ts +11 -29
- package/dist/task/shipped-source.js +11 -29
- package/dist/task/skip-escape.js +10 -14
- package/dist/task/spec-urls.d.ts +26 -65
- package/dist/task/spec-urls.js +26 -65
- package/dist/task/spec-validation.d.ts +17 -20
- package/dist/task/spec-validation.js +17 -20
- package/dist/task/stall-detector.d.ts +23 -30
- package/dist/task/stall-detector.js +23 -30
- package/dist/task/stream-watchdog.d.ts +14 -12
- package/dist/task/stream-watchdog.js +14 -12
- package/dist/task/substitution-probe.d.ts +17 -20
- package/dist/task/substitution-probe.js +17 -20
- package/dist/task/task-gates.d.ts +36 -41
- package/dist/task/task-gates.js +95 -106
- package/dist/task/task-io.d.ts +4 -4
- package/dist/task/task-io.js +4 -4
- package/dist/task/task-parsers.js +4 -3
- package/dist/task/task-provenance.d.ts +2 -2
- package/dist/task/task-provenance.js +11 -13
- package/dist/task/task-types.d.ts +4 -3
- package/dist/task/terminal-outcome.d.ts +14 -16
- package/dist/task/terminal-outcome.js +12 -14
- package/dist/task/test-assembly.d.ts +13 -20
- package/dist/task/test-assembly.js +13 -20
- package/dist/task/timings.d.ts +5 -3
- package/dist/task/timings.js +5 -3
- package/dist/task/title-label.d.ts +9 -4
- package/dist/task/title-label.js +9 -4
- package/dist/task/type-only-answer.d.ts +44 -52
- package/dist/task/type-only-answer.js +44 -52
- package/dist/task/unfailable-command.d.ts +18 -24
- package/dist/task/unfailable-command.js +21 -27
- package/dist/task/unknown-routing.d.ts +10 -4
- package/dist/task/unknown-routing.js +10 -4
- package/dist/task/user-directives.d.ts +5 -8
- package/dist/task/user-directives.js +5 -8
- package/dist/task/verify-quality.d.ts +18 -22
- package/dist/task/verify-quality.js +45 -46
- package/dist/task/verify-reconcile.d.ts +15 -10
- package/dist/task/verify-reconcile.js +45 -43
- package/dist/task/verify-resolution.d.ts +24 -20
- package/dist/task/verify-resolution.js +51 -50
- package/dist/task/verify-work.d.ts +59 -66
- package/dist/task/verify-work.js +101 -138
- package/dist/task/widget.d.ts +15 -14
- package/dist/task/widget.js +22 -17
- package/dist/task/wiring-claims.d.ts +25 -32
- package/dist/task/wiring-claims.js +30 -35
- package/dist/task/write-guard.d.ts +39 -39
- package/dist/task/write-guard.js +48 -51
- package/dist/task/yolo.d.ts +34 -30
- package/dist/task/yolo.js +42 -37
- package/dist/workers/abstention.d.ts +21 -41
- package/dist/workers/abstention.js +27 -48
- package/dist/workers/brave-search.d.ts +4 -3
- package/dist/workers/brave-search.js +5 -2
- package/dist/workers/brave-warning.d.ts +7 -4
- package/dist/workers/brave-warning.js +19 -7
- package/dist/workers/ddg-search.d.ts +6 -6
- package/dist/workers/ddg-search.js +18 -12
- package/dist/workers/docs-cache.js +5 -2
- package/dist/workers/docs-chunk.d.ts +30 -37
- package/dist/workers/docs-chunk.js +37 -41
- package/dist/workers/docs-core.d.ts +28 -44
- package/dist/workers/docs-core.js +25 -44
- package/dist/workers/docs-index.js +4 -3
- package/dist/workers/docs-lookup.d.ts +15 -22
- package/dist/workers/docs-lookup.js +12 -21
- package/dist/workers/docs-project.d.ts +15 -9
- package/dist/workers/docs-project.js +17 -10
- package/dist/workers/docs-resolve.d.ts +19 -20
- package/dist/workers/docs-resolve.js +35 -32
- package/dist/workers/docs-retrieve.d.ts +5 -6
- package/dist/workers/docs-retrieve.js +18 -15
- package/dist/workers/exa-search.d.ts +9 -6
- package/dist/workers/exa-search.js +23 -12
- package/dist/workers/fetch-core.d.ts +13 -16
- package/dist/workers/fetch-core.js +23 -23
- package/dist/workers/focused-extractor.d.ts +12 -12
- package/dist/workers/focused-extractor.js +16 -19
- package/dist/workers/html-clean.js +24 -14
- package/dist/workers/http-request.d.ts +28 -20
- package/dist/workers/http-request.js +22 -17
- package/dist/workers/npm-version.d.ts +28 -11
- package/dist/workers/npm-version.js +24 -15
- package/dist/workers/phantom-imports.d.ts +15 -12
- package/dist/workers/phantom-imports.js +30 -24
- package/dist/workers/pi-worker-core.d.ts +69 -71
- package/dist/workers/pi-worker-core.js +100 -109
- package/dist/workers/pi-worker-docs.d.ts +24 -19
- package/dist/workers/pi-worker-docs.js +67 -76
- package/dist/workers/pi-worker-fetch.d.ts +7 -3
- package/dist/workers/pi-worker-fetch.js +27 -19
- package/dist/workers/pi-worker-search.js +12 -8
- package/dist/workers/pi-worker.d.ts +9 -4
- package/dist/workers/pi-worker.js +21 -14
- package/dist/workers/reasoning-warning.d.ts +18 -17
- package/dist/workers/reasoning-warning.js +22 -20
- package/dist/workers/research-cache.js +50 -78
- package/dist/workers/search-core.js +7 -5
- package/dist/workers/search-types.d.ts +10 -9
- package/dist/workers/search-types.js +9 -8
- package/dist/workers/session-hint.d.ts +13 -14
- package/dist/workers/session-hint.js +8 -9
- package/dist/workers/shared.d.ts +21 -25
- package/dist/workers/shared.js +0 -0
- package/dist/workers/single-read-extension.d.ts +14 -7
- package/dist/workers/single-read-extension.js +14 -7
- package/dist/workers/single-read-guard.d.ts +25 -28
- package/dist/workers/single-read-guard.js +32 -32
- package/dist/workers/typeonly-log.d.ts +12 -9
- package/dist/workers/typeonly-log.js +29 -33
- package/dist/workers/worker-channels.d.ts +15 -23
- package/dist/workers/worker-channels.js +15 -23
- package/dist/workers/worker-failure.d.ts +38 -46
- package/dist/workers/worker-failure.js +31 -39
- package/dist/workers/worker-kill.d.ts +25 -26
- package/dist/workers/worker-kill.js +16 -19
- package/dist/workers/worker-profiles.d.ts +43 -53
- package/dist/workers/worker-profiles.js +30 -38
- package/package.json +10 -8
|
@@ -13,15 +13,15 @@ export function parseVerifyBlock(spec) {
|
|
|
13
13
|
/**
|
|
14
14
|
* parseVerifyBlock, but only when the fenced block is actually CLOSED.
|
|
15
15
|
*
|
|
16
|
-
* An unterminated fence makes the lenient parser swallow the rest of the file:
|
|
17
|
-
*
|
|
18
|
-
*
|
|
16
|
+
* An unterminated fence makes the lenient parser swallow the rest of the file: a
|
|
17
|
+
* task file that opens ```sh and never closes it yields "VERIFY commands" that
|
|
18
|
+
* include every line appended after the spec — a timings table, a gate-trail line.
|
|
19
19
|
* That is harmless where the parser only asks "is there something runnable here",
|
|
20
|
-
* and NOT harmless where a parsed line is treated as
|
|
21
|
-
* quoting `bun run lint` would match a
|
|
20
|
+
* and NOT harmless where a parsed line is treated as PROVENANCE: a debt reason
|
|
21
|
+
* quoting `bun run lint` would then match a trail sentence and mint a stored,
|
|
22
22
|
* re-runnable command the spec never asked for (accept-debt.ts
|
|
23
|
-
* verifyCommandFromReason
|
|
24
|
-
* to MEAN something use this one: an unclosed fence is no block at all.
|
|
23
|
+
* `verifyCommandFromReason`, `inv-command-provenance`). Callers that need the
|
|
24
|
+
* block to MEAN something use this one: an unclosed fence is no block at all.
|
|
25
25
|
*/
|
|
26
26
|
export function parseVerifyBlockStrict(spec) {
|
|
27
27
|
const scan = scanVerifyBlock(spec);
|
|
@@ -119,24 +119,21 @@ export const REFINE_SECTIONS = [
|
|
|
119
119
|
/**
|
|
120
120
|
* Is a refine child's output shaped like a refined prompt?
|
|
121
121
|
*
|
|
122
|
-
*
|
|
123
|
-
*
|
|
124
|
-
*
|
|
125
|
-
* `
|
|
126
|
-
*
|
|
127
|
-
*
|
|
128
|
-
* once and says nothing. This names the contract in one place.
|
|
122
|
+
* Every downstream reader of a refined prompt is a PARTIAL parser that tolerates a
|
|
123
|
+
* missing section SILENTLY: `extractCapsSection` (refuted-constraint.ts) returns
|
|
124
|
+
* null, `scopedToolingGoal` (phases.ts) returns the whole text, and `deriveTitle`
|
|
125
|
+
* (parsers.ts) and `extractEnrichTargets` (enrichment.ts) fall back. So a refine
|
|
126
|
+
* answer that dropped a heading degrades four features at once and says nothing.
|
|
127
|
+
* This names the contract in one place; phases.ts calls it after refine.
|
|
129
128
|
*
|
|
130
129
|
* WHAT IT CHECKS, and why it is only this. Every one of those consumers looks
|
|
131
130
|
* for a BARE ALL-CAPS heading alone on its own line — `l.trim() === heading`,
|
|
132
131
|
* `/^GOAL[ \t]*\n/m`. That is the operative contract, so that is the test.
|
|
133
132
|
*
|
|
134
|
-
* It does NOT require the text to START with GOAL, even though REFINE_PROMPT
|
|
135
|
-
*
|
|
136
|
-
*
|
|
137
|
-
*
|
|
138
|
-
* refine output usually looks like, and production has always consumed it fine.
|
|
139
|
-
* A validator stricter than its consumers would reject work that works.
|
|
133
|
+
* It does NOT require the text to START with GOAL, even though REFINE_PROMPT asks
|
|
134
|
+
* for the four headings in order and forbids a preamble. None of the four
|
|
135
|
+
* consumers above cares where the heading sits, so a validator that did would
|
|
136
|
+
* reject work every one of them handles.
|
|
140
137
|
*
|
|
141
138
|
* Returns a problem string, or null when the shape is good — same contract as
|
|
142
139
|
* `validateSpecShape` above.
|
|
@@ -2,16 +2,11 @@
|
|
|
2
2
|
* Progress-based runaway guard for phase children — the replacement for a
|
|
3
3
|
* wall-clock cap.
|
|
4
4
|
*
|
|
5
|
-
* WHY NOT SECONDS.
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
* captured auto-decompose request, everything else byte-identical): every single
|
|
11
|
-
* healthy run took 610-927s and produced 26-42 correct titles. The cap would
|
|
12
|
-
* have killed 10 out of 10 GOOD runs. A slower model, a bigger design doc or a
|
|
13
|
-
* longer reasoning budget moves that number again — so any constant in seconds
|
|
14
|
-
* is wrong for someone.
|
|
5
|
+
* WHY NOT SECONDS. A wall clock has to be sized against how long a HEALTHY child
|
|
6
|
+
* takes, and that is not a property of the pathology — it is a property of the
|
|
7
|
+
* model, its sampler settings, the reasoning budget and the size of the document
|
|
8
|
+
* being read. Any of those moving turns the margin into a killer of good runs.
|
|
9
|
+
* `PHASE_CHILD_TIMEOUT_MS` is 0 (off) for exactly that reason.
|
|
15
10
|
*
|
|
16
11
|
* WHAT REPLACES IT. Two bounds, both dimensionless — invariant to model speed,
|
|
17
12
|
* project size and reasoning budget:
|
|
@@ -22,26 +17,25 @@
|
|
|
22
17
|
* never trips, however slow it is. A child re-opening the same four files
|
|
23
18
|
* trips after NO_PROGRESS_LIMIT of them, however fast it is.
|
|
24
19
|
*
|
|
25
|
-
* Judged on the RESULT, not the arguments, because the arguments lie.
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
* refusals in a row, which is what it actually was.
|
|
20
|
+
* Judged on the RESULT, not the arguments, because the arguments lie. A child
|
|
21
|
+
* whose reads are all REFUSED — by the single-read guard, say — can issue
|
|
22
|
+
* them at steadily rising offsets, so by arguments it looks like textbook
|
|
23
|
+
* forward paging and an offset rule waves every one through. By result it is
|
|
24
|
+
* one identical refusal after another.
|
|
31
25
|
*
|
|
32
26
|
* 2. CONTEXT CHURN. Sum the bytes of tool RESULTS the child has pulled in. Once
|
|
33
27
|
* that exceeds CONTEXT_CHURN_FACTOR times its own context window and it
|
|
34
28
|
* still has not answered, it has necessarily forgotten what it read first
|
|
35
|
-
* and is re-reading to fill a window pi keeps compacting.
|
|
36
|
-
*
|
|
37
|
-
*
|
|
38
|
-
* rule 1 alone would not have caught it. The bound scales with the model's
|
|
29
|
+
* and is re-reading to fill a window pi keeps compacting. This catches the
|
|
30
|
+
* child rule 1 cannot: one that pages FORWARD the whole time, pulling in new
|
|
31
|
+
* bytes on every call, and never converges. The bound scales with the model's
|
|
39
32
|
* OWN window, so a 1M-context model gets a 1M-context allowance.
|
|
40
33
|
*
|
|
41
34
|
* The window is supplied by the PARENT at spawn (`PhaseDeps.contextWindow`),
|
|
42
|
-
* not read off the child's event stream: pi
|
|
43
|
-
*
|
|
44
|
-
*
|
|
35
|
+
* not read off the child's event stream: pi's `--mode json` stream carries no
|
|
36
|
+
* context event at all. A detector that waited to be told one sits at 0, and
|
|
37
|
+
* `churnTripped` returns false on a non-positive window — so the rule would
|
|
38
|
+
* never fire.
|
|
45
39
|
*
|
|
46
40
|
* Neither rule can fire on a child that is thinking rather than calling tools:
|
|
47
41
|
* that case is bounded by the model's max tokens (server-enforced) and by the
|
|
@@ -58,11 +52,10 @@ import type { LoopHit, ToolCall } from '../shared/child-process.js';
|
|
|
58
52
|
*
|
|
59
53
|
* Eight, because the honest reasons to get back something you have already seen
|
|
60
54
|
* are few and bounded: re-checking a file after an edit, a grep that lands in a
|
|
61
|
-
* file already read, a retry after a malformed call, a missing path. A child
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
* is wide enough that the exact value is not load-bearing.
|
|
55
|
+
* file already read, a retry after a malformed call, a missing path. A child doing
|
|
56
|
+
* real work interleaves those with progress and RESETS the counter, so the streak
|
|
57
|
+
* only survives when nothing at all is being learned. The exact value is not
|
|
58
|
+
* load-bearing: a genuine thrash never resets and runs on indefinitely.
|
|
66
59
|
*/
|
|
67
60
|
export declare const NO_PROGRESS_LIMIT = 8;
|
|
68
61
|
/**
|
|
@@ -109,7 +102,7 @@ export declare class StallDetector {
|
|
|
109
102
|
/**
|
|
110
103
|
* Restart hint for a child killed by the stall detector. Names the specific
|
|
111
104
|
* mistake — re-reading covered ground vs pulling in more than it can hold —
|
|
112
|
-
* because "you ran out of time"
|
|
113
|
-
*
|
|
105
|
+
* because "you ran out of time" tells a model that was working correctly but
|
|
106
|
+
* slowly to truncate its work for no reason.
|
|
114
107
|
*/
|
|
115
108
|
export declare function formatStallHint(kind: StallKind): string;
|
|
@@ -2,16 +2,11 @@
|
|
|
2
2
|
* Progress-based runaway guard for phase children — the replacement for a
|
|
3
3
|
* wall-clock cap.
|
|
4
4
|
*
|
|
5
|
-
* WHY NOT SECONDS.
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
* captured auto-decompose request, everything else byte-identical): every single
|
|
11
|
-
* healthy run took 610-927s and produced 26-42 correct titles. The cap would
|
|
12
|
-
* have killed 10 out of 10 GOOD runs. A slower model, a bigger design doc or a
|
|
13
|
-
* longer reasoning budget moves that number again — so any constant in seconds
|
|
14
|
-
* is wrong for someone.
|
|
5
|
+
* WHY NOT SECONDS. A wall clock has to be sized against how long a HEALTHY child
|
|
6
|
+
* takes, and that is not a property of the pathology — it is a property of the
|
|
7
|
+
* model, its sampler settings, the reasoning budget and the size of the document
|
|
8
|
+
* being read. Any of those moving turns the margin into a killer of good runs.
|
|
9
|
+
* `PHASE_CHILD_TIMEOUT_MS` is 0 (off) for exactly that reason.
|
|
15
10
|
*
|
|
16
11
|
* WHAT REPLACES IT. Two bounds, both dimensionless — invariant to model speed,
|
|
17
12
|
* project size and reasoning budget:
|
|
@@ -22,26 +17,25 @@
|
|
|
22
17
|
* never trips, however slow it is. A child re-opening the same four files
|
|
23
18
|
* trips after NO_PROGRESS_LIMIT of them, however fast it is.
|
|
24
19
|
*
|
|
25
|
-
* Judged on the RESULT, not the arguments, because the arguments lie.
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
* refusals in a row, which is what it actually was.
|
|
20
|
+
* Judged on the RESULT, not the arguments, because the arguments lie. A child
|
|
21
|
+
* whose reads are all REFUSED — by the single-read guard, say — can issue
|
|
22
|
+
* them at steadily rising offsets, so by arguments it looks like textbook
|
|
23
|
+
* forward paging and an offset rule waves every one through. By result it is
|
|
24
|
+
* one identical refusal after another.
|
|
31
25
|
*
|
|
32
26
|
* 2. CONTEXT CHURN. Sum the bytes of tool RESULTS the child has pulled in. Once
|
|
33
27
|
* that exceeds CONTEXT_CHURN_FACTOR times its own context window and it
|
|
34
28
|
* still has not answered, it has necessarily forgotten what it read first
|
|
35
|
-
* and is re-reading to fill a window pi keeps compacting.
|
|
36
|
-
*
|
|
37
|
-
*
|
|
38
|
-
* rule 1 alone would not have caught it. The bound scales with the model's
|
|
29
|
+
* and is re-reading to fill a window pi keeps compacting. This catches the
|
|
30
|
+
* child rule 1 cannot: one that pages FORWARD the whole time, pulling in new
|
|
31
|
+
* bytes on every call, and never converges. The bound scales with the model's
|
|
39
32
|
* OWN window, so a 1M-context model gets a 1M-context allowance.
|
|
40
33
|
*
|
|
41
34
|
* The window is supplied by the PARENT at spawn (`PhaseDeps.contextWindow`),
|
|
42
|
-
* not read off the child's event stream: pi
|
|
43
|
-
*
|
|
44
|
-
*
|
|
35
|
+
* not read off the child's event stream: pi's `--mode json` stream carries no
|
|
36
|
+
* context event at all. A detector that waited to be told one sits at 0, and
|
|
37
|
+
* `churnTripped` returns false on a non-positive window — so the rule would
|
|
38
|
+
* never fire.
|
|
45
39
|
*
|
|
46
40
|
* Neither rule can fire on a child that is thinking rather than calling tools:
|
|
47
41
|
* that case is bounded by the model's max tokens (server-enforced) and by the
|
|
@@ -58,11 +52,10 @@ import { stableStringify } from './loop-detector.js';
|
|
|
58
52
|
*
|
|
59
53
|
* Eight, because the honest reasons to get back something you have already seen
|
|
60
54
|
* are few and bounded: re-checking a file after an edit, a grep that lands in a
|
|
61
|
-
* file already read, a retry after a malformed call, a missing path. A child
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
* is wide enough that the exact value is not load-bearing.
|
|
55
|
+
* file already read, a retry after a malformed call, a missing path. A child doing
|
|
56
|
+
* real work interleaves those with progress and RESETS the counter, so the streak
|
|
57
|
+
* only survives when nothing at all is being learned. The exact value is not
|
|
58
|
+
* load-bearing: a genuine thrash never resets and runs on indefinitely.
|
|
66
59
|
*/
|
|
67
60
|
export const NO_PROGRESS_LIMIT = 8;
|
|
68
61
|
/**
|
|
@@ -146,8 +139,8 @@ export class StallDetector {
|
|
|
146
139
|
/**
|
|
147
140
|
* Restart hint for a child killed by the stall detector. Names the specific
|
|
148
141
|
* mistake — re-reading covered ground vs pulling in more than it can hold —
|
|
149
|
-
* because "you ran out of time"
|
|
150
|
-
*
|
|
142
|
+
* because "you ran out of time" tells a model that was working correctly but
|
|
143
|
+
* slowly to truncate its work for no reason.
|
|
151
144
|
*/
|
|
152
145
|
export function formatStallHint(kind) {
|
|
153
146
|
if (kind === 'context-churn') {
|
|
@@ -2,12 +2,12 @@ import type { ExtensionAPI } from '@earendil-works/pi-coding-agent';
|
|
|
2
2
|
/**
|
|
3
3
|
* MAIN-SESSION adapter for the model-stream watchdog.
|
|
4
4
|
*
|
|
5
|
-
* WHY
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
5
|
+
* WHY: a turn can die mid-stream — the last thing recorded is an ordinary
|
|
6
|
+
* assistant message, then silence forever, with the model endpoint still up.
|
|
7
|
+
* Nothing throws for that shape, so the connection-error retry (which needs a
|
|
8
|
+
* reported ModelError) cannot fire, and the command watchdog only covers tool
|
|
9
|
+
* executions, so it never arms. Without this the session sits dead until a human
|
|
10
|
+
* notices.
|
|
11
11
|
*
|
|
12
12
|
* HOW: pi's extension events ARE the stream. Any of them — a token delta, a
|
|
13
13
|
* thinking delta, a tool-call delta, the provider's response headers — resets the
|
|
@@ -17,16 +17,18 @@ import type { ExtensionAPI } from '@earendil-works/pi-coding-agent';
|
|
|
17
17
|
* blind re-send would re-run them).
|
|
18
18
|
*
|
|
19
19
|
* ONE ABORT CHANNEL: the fire path goes through the command watchdog's existing
|
|
20
|
-
* {@link noteWatchdogAbort} flag and its WATCHDOG_CANCEL_MARKER, so
|
|
21
|
-
* steerUntilDone
|
|
22
|
-
*
|
|
20
|
+
* {@link noteWatchdogAbort} flag and its WATCHDOG_CANCEL_MARKER, so the abort/steer
|
|
21
|
+
* race steerUntilDone already handles covers this watchdog too, instead of racing a
|
|
22
|
+
* second, parallel abort mechanism.
|
|
23
23
|
*
|
|
24
24
|
* SUSPENDED DURING TOOLS: while a tool executes the model stream is legitimately
|
|
25
|
-
* idle — a
|
|
25
|
+
* idle — a long build emits nothing on it. That window belongs to the command
|
|
26
26
|
* watchdog (requestTimeoutMs); this one pauses between tool_execution_start and
|
|
27
27
|
* tool_execution_end so the two can never double-fire on the same silence.
|
|
28
28
|
*
|
|
29
|
-
* SCOPE: main session only. Children run `--no-extensions
|
|
30
|
-
*
|
|
29
|
+
* SCOPE: main session only. Children run `--no-extensions` (CHILD_BASE_ARGS), so
|
|
30
|
+
* none of these events reaches them; their equivalent guard is a second
|
|
31
|
+
* `StreamWatchdog` inside runChild (shared/child-process.ts), on the same
|
|
32
|
+
* machine.
|
|
31
33
|
*/
|
|
32
34
|
export declare function registerStreamWatchdog(pi: ExtensionAPI): void;
|
|
@@ -4,12 +4,12 @@ import { noteWatchdogAbort, WATCHDOG_CANCEL_MARKER } from './command-watchdog.js
|
|
|
4
4
|
/**
|
|
5
5
|
* MAIN-SESSION adapter for the model-stream watchdog.
|
|
6
6
|
*
|
|
7
|
-
* WHY
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
7
|
+
* WHY: a turn can die mid-stream — the last thing recorded is an ordinary
|
|
8
|
+
* assistant message, then silence forever, with the model endpoint still up.
|
|
9
|
+
* Nothing throws for that shape, so the connection-error retry (which needs a
|
|
10
|
+
* reported ModelError) cannot fire, and the command watchdog only covers tool
|
|
11
|
+
* executions, so it never arms. Without this the session sits dead until a human
|
|
12
|
+
* notices.
|
|
13
13
|
*
|
|
14
14
|
* HOW: pi's extension events ARE the stream. Any of them — a token delta, a
|
|
15
15
|
* thinking delta, a tool-call delta, the provider's response headers — resets the
|
|
@@ -19,17 +19,19 @@ import { noteWatchdogAbort, WATCHDOG_CANCEL_MARKER } from './command-watchdog.js
|
|
|
19
19
|
* blind re-send would re-run them).
|
|
20
20
|
*
|
|
21
21
|
* ONE ABORT CHANNEL: the fire path goes through the command watchdog's existing
|
|
22
|
-
* {@link noteWatchdogAbort} flag and its WATCHDOG_CANCEL_MARKER, so
|
|
23
|
-
* steerUntilDone
|
|
24
|
-
*
|
|
22
|
+
* {@link noteWatchdogAbort} flag and its WATCHDOG_CANCEL_MARKER, so the abort/steer
|
|
23
|
+
* race steerUntilDone already handles covers this watchdog too, instead of racing a
|
|
24
|
+
* second, parallel abort mechanism.
|
|
25
25
|
*
|
|
26
26
|
* SUSPENDED DURING TOOLS: while a tool executes the model stream is legitimately
|
|
27
|
-
* idle — a
|
|
27
|
+
* idle — a long build emits nothing on it. That window belongs to the command
|
|
28
28
|
* watchdog (requestTimeoutMs); this one pauses between tool_execution_start and
|
|
29
29
|
* tool_execution_end so the two can never double-fire on the same silence.
|
|
30
30
|
*
|
|
31
|
-
* SCOPE: main session only. Children run `--no-extensions
|
|
32
|
-
*
|
|
31
|
+
* SCOPE: main session only. Children run `--no-extensions` (CHILD_BASE_ARGS), so
|
|
32
|
+
* none of these events reaches them; their equivalent guard is a second
|
|
33
|
+
* `StreamWatchdog` inside runChild (shared/child-process.ts), on the same
|
|
34
|
+
* machine.
|
|
33
35
|
*/
|
|
34
36
|
export function registerStreamWatchdog(pi) {
|
|
35
37
|
// The ctx whose abort() ends the in-flight turn, refreshed on every event so
|
|
@@ -2,19 +2,15 @@
|
|
|
2
2
|
* substitution-probe — deterministic detection of SELF-VERIFIED work, feeding the
|
|
3
3
|
* verify gate's prompt.
|
|
4
4
|
*
|
|
5
|
-
* The failure class
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
* The suite runs 26/26 green while every protected route of the real server 500s.
|
|
11
|
-
* The verify child judged "do the tests pass", saw green, and PASSed.
|
|
5
|
+
* The failure class: an implementation that hits a real bug in the shipped app
|
|
6
|
+
* can ship "integration tests" that re-implement the API inline instead — a test
|
|
7
|
+
* file that stands up its own server, or imports the real app and never calls it.
|
|
8
|
+
* The suite then runs green while every route of the real server fails, and a
|
|
9
|
+
* verify child that asks "do the tests pass" sees green and PASSes.
|
|
12
10
|
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* rule PLUS a probe finding naming the suspect file caught 5/5, each verdict naming
|
|
17
|
-
* the exact inline handlers and confirming the real app crashes.
|
|
11
|
+
* The prompt rule alone is weak against that, because the child has to notice the
|
|
12
|
+
* substitution unprompted. This probe hands it a concrete finding naming the
|
|
13
|
+
* suspect file, so the rule fires on a line rather than on self-discovery.
|
|
18
14
|
*
|
|
19
15
|
* WHAT THE PROBE ASSERTS is deliberately the class INVARIANT, not any framework
|
|
20
16
|
* shape: when a task's own diff includes the very tests whose green result would
|
|
@@ -23,15 +19,16 @@
|
|
|
23
19
|
* That is computable from pure git diff shape (which changed files live in test
|
|
24
20
|
* paths, and how many lines the task added there): no language parsing, no server-
|
|
25
21
|
* constructor lists, no import syntax — a Python, Go, or Rust project produces the
|
|
26
|
-
* same finding the same way.
|
|
27
|
-
* constructors; that only re-encoded the one incident. A behavioral canary — rerun
|
|
28
|
-
* the suite with shipped sources removed, still-green ⇒ copy — was also rejected:
|
|
29
|
-
* the real mx5 copy still imported the shipped db module, so poisoning sources
|
|
30
|
-
* breaks the copy's tests too and the conviction never fires.)
|
|
22
|
+
* same finding the same way.
|
|
31
23
|
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
34
|
-
*
|
|
24
|
+
* A behavioural canary — re-run the suite with the shipped sources removed, and
|
|
25
|
+
* treat still-green as proof of a copy — does NOT work here: a substituted test
|
|
26
|
+
* usually still imports SOMETHING real (a db module, a type), so poisoning the
|
|
27
|
+
* sources breaks it too and the canary never convicts.
|
|
28
|
+
*
|
|
29
|
+
* Findings are advisory: they mandate the spot-check, they never auto-FAIL. An
|
|
30
|
+
* honest self-authored test survives that spot-check; only one that bypasses the
|
|
31
|
+
* artifact gets named in a FAIL.
|
|
35
32
|
*/
|
|
36
33
|
/** One changed file as the git-shape collector reports it. */
|
|
37
34
|
export interface ChangedFile {
|
|
@@ -2,19 +2,15 @@
|
|
|
2
2
|
* substitution-probe — deterministic detection of SELF-VERIFIED work, feeding the
|
|
3
3
|
* verify gate's prompt.
|
|
4
4
|
*
|
|
5
|
-
* The failure class
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
* The suite runs 26/26 green while every protected route of the real server 500s.
|
|
11
|
-
* The verify child judged "do the tests pass", saw green, and PASSed.
|
|
5
|
+
* The failure class: an implementation that hits a real bug in the shipped app
|
|
6
|
+
* can ship "integration tests" that re-implement the API inline instead — a test
|
|
7
|
+
* file that stands up its own server, or imports the real app and never calls it.
|
|
8
|
+
* The suite then runs green while every route of the real server fails, and a
|
|
9
|
+
* verify child that asks "do the tests pass" sees green and PASSes.
|
|
12
10
|
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* rule PLUS a probe finding naming the suspect file caught 5/5, each verdict naming
|
|
17
|
-
* the exact inline handlers and confirming the real app crashes.
|
|
11
|
+
* The prompt rule alone is weak against that, because the child has to notice the
|
|
12
|
+
* substitution unprompted. This probe hands it a concrete finding naming the
|
|
13
|
+
* suspect file, so the rule fires on a line rather than on self-discovery.
|
|
18
14
|
*
|
|
19
15
|
* WHAT THE PROBE ASSERTS is deliberately the class INVARIANT, not any framework
|
|
20
16
|
* shape: when a task's own diff includes the very tests whose green result would
|
|
@@ -23,15 +19,16 @@
|
|
|
23
19
|
* That is computable from pure git diff shape (which changed files live in test
|
|
24
20
|
* paths, and how many lines the task added there): no language parsing, no server-
|
|
25
21
|
* constructor lists, no import syntax — a Python, Go, or Rust project produces the
|
|
26
|
-
* same finding the same way.
|
|
27
|
-
* constructors; that only re-encoded the one incident. A behavioral canary — rerun
|
|
28
|
-
* the suite with shipped sources removed, still-green ⇒ copy — was also rejected:
|
|
29
|
-
* the real mx5 copy still imported the shipped db module, so poisoning sources
|
|
30
|
-
* breaks the copy's tests too and the conviction never fires.)
|
|
22
|
+
* same finding the same way.
|
|
31
23
|
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
34
|
-
*
|
|
24
|
+
* A behavioural canary — re-run the suite with the shipped sources removed, and
|
|
25
|
+
* treat still-green as proof of a copy — does NOT work here: a substituted test
|
|
26
|
+
* usually still imports SOMETHING real (a db module, a type), so poisoning the
|
|
27
|
+
* sources breaks it too and the canary never convicts.
|
|
28
|
+
*
|
|
29
|
+
* Findings are advisory: they mandate the spot-check, they never auto-FAIL. An
|
|
30
|
+
* honest self-authored test survives that spot-check; only one that bypasses the
|
|
31
|
+
* artifact gets named in a FAIL.
|
|
35
32
|
*/
|
|
36
33
|
/** Is this a file whose job is testing — by suffix or by living in a test dir? */
|
|
37
34
|
export function isTestFile(p) {
|
|
@@ -69,10 +69,8 @@ export interface GateDeps {
|
|
|
69
69
|
* it has always been, so production wiring is untouched.
|
|
70
70
|
*
|
|
71
71
|
* Its own field because it answers a different question from the gate above.
|
|
72
|
-
*
|
|
73
|
-
*
|
|
74
|
-
* solely to unlock `mode === 'edit'`, re-invented in the suite and again in
|
|
75
|
-
* scripts/enforce-revert-attribution-replay-ab.ts.
|
|
72
|
+
* Sharing one field would leave a caller — or a test — no way to answer the two
|
|
73
|
+
* differently except by counting invocations.
|
|
76
74
|
*/
|
|
77
75
|
reVerify?: (ctx: ExtensionCommandContext, cwd: string, taskTitle: string, taskId: string) => Promise<VerifyOutcome>;
|
|
78
76
|
/**
|
|
@@ -96,10 +94,10 @@ export interface GateDeps {
|
|
|
96
94
|
/**
|
|
97
95
|
* BOUNDED fix for a repo-health verify FAIL: a small read,edit,bash child fixes
|
|
98
96
|
* exactly the static findings (revert-guarded — see lint-fix.ts), instead of the
|
|
99
|
-
* full implementation re-run AUTOFIX reaches for.
|
|
100
|
-
*
|
|
101
|
-
*
|
|
102
|
-
*
|
|
97
|
+
* full implementation re-run AUTOFIX reaches for. Smallest tool first: a static
|
|
98
|
+
* finding does not need the whole turn re-run to fix it. Runs at most once per
|
|
99
|
+
* gate sequence; not-applied falls through to the picker. Absent → the loop goes
|
|
100
|
+
* straight to recommend/picker.
|
|
103
101
|
*/
|
|
104
102
|
lintFix?: (ctx: ExtensionCommandContext, cwd: string, taskTitle: string, taskId: string, failReason: string) => Promise<{
|
|
105
103
|
ok: boolean;
|
|
@@ -107,15 +105,15 @@ export interface GateDeps {
|
|
|
107
105
|
}>;
|
|
108
106
|
/**
|
|
109
107
|
* Deterministic whole-repo static check (repo-health), used as the PRE-COMMIT
|
|
110
|
-
* gate on an edit-mode enforce pass:
|
|
111
|
-
*
|
|
112
|
-
*
|
|
113
|
-
* commit-then-differential path runs
|
|
108
|
+
* gate on an edit-mode enforce pass: an enforce edit that breaks the project's
|
|
109
|
+
* own lint would otherwise cost a commit, a model re-verify and a revert before
|
|
110
|
+
* anything noticed. Checking first skips that cycle. Absent → the
|
|
111
|
+
* commit-then-differential path runs on its own.
|
|
112
|
+
*
|
|
113
|
+
* Takes the live ctx and a label so the implementation can render a status line
|
|
114
|
+
* while it runs: it is as slow as the project's own lint, and a gate step that
|
|
115
|
+
* long with no widget is indistinguishable from a hang.
|
|
114
116
|
*/
|
|
115
|
-
/** Deterministic whole-repo static check for the enforce pre-commit gate. Takes
|
|
116
|
-
* the live ctx and a label so the implementation can render a status line while
|
|
117
|
-
* it runs — it is as slow as the project's own lint (15–69s measured), and a
|
|
118
|
-
* gate step that long with no widget is indistinguishable from a hang. */
|
|
119
117
|
repoHealth?: (ctx: ExtensionCommandContext, cwd: string, label: string) => Promise<{
|
|
120
118
|
ok: boolean;
|
|
121
119
|
reason: string;
|
|
@@ -131,10 +129,10 @@ export interface GateDeps {
|
|
|
131
129
|
* Append one line to the task's durable gate trail (`## gates` in the task
|
|
132
130
|
* file). Every gate outcome — each verify verdict, the user's FAIL resolution,
|
|
133
131
|
* the commit result, enforce mode + verdict, the differential guard's decision —
|
|
134
|
-
* is recorded so the sequence is auditable from artifacts alone
|
|
135
|
-
*
|
|
136
|
-
*
|
|
137
|
-
*
|
|
132
|
+
* is recorded so the sequence is auditable from artifacts alone. A verdict that
|
|
133
|
+
* lives only in a terminal notify cannot answer "why did enforce not run here?"
|
|
134
|
+
* after the fact. Best-effort: absent in tests → skipped; failures are swallowed
|
|
135
|
+
* by the implementation, never by this sequence.
|
|
138
136
|
*/
|
|
139
137
|
record?: (cwd: string, taskId: string, line: string) => Promise<void>;
|
|
140
138
|
/**
|
|
@@ -172,8 +170,8 @@ export interface GateDeps {
|
|
|
172
170
|
* verify site); `committed` = the files the task snapshot + the ENFORCE commit
|
|
173
171
|
* changed (the post-commit enforce site); `enforce-commit` = the ENFORCE COMMIT
|
|
174
172
|
* ALONE, which is the only correct authorship question at the enforce
|
|
175
|
-
* differential
|
|
176
|
-
*
|
|
173
|
+
* differential — that differential decides whether to discard the ENFORCE
|
|
174
|
+
* COMMIT, so what the TASK touched is irrelevant to it.
|
|
177
175
|
* `null` means UNKNOWN (git unavailable) and stands the whole channel down —
|
|
178
176
|
* inconclusive is never evidence, so an unreadable tree can only cost a repair
|
|
179
177
|
* task, never spawn a wrong one or wrongly keep a regression.
|
|
@@ -200,8 +198,9 @@ export interface GateDeps {
|
|
|
200
198
|
/**
|
|
201
199
|
* Restore the given frozen paths to their committed (HEAD) state, discarding a
|
|
202
200
|
* gate child's edits to them, and return the files actually reverted. Prompt
|
|
203
|
-
* framing is
|
|
204
|
-
* the write is undone, not merely warned about. Absent → the
|
|
201
|
+
* framing is insufficient for this class (see frozen-path-guard.ts), so the deny
|
|
202
|
+
* is mechanical: the write is undone, not merely warned about. Absent → the
|
|
203
|
+
* guard warns only.
|
|
205
204
|
*/
|
|
206
205
|
revertFrozenPaths?: (cwd: string, paths: string[]) => Promise<string[]>;
|
|
207
206
|
}
|
|
@@ -286,13 +285,12 @@ export interface YoloAcceptContext {
|
|
|
286
285
|
/**
|
|
287
286
|
* The reason an auto-ACCEPT is being written — NAMED, not assumed.
|
|
288
287
|
*
|
|
289
|
-
*
|
|
290
|
-
*
|
|
291
|
-
*
|
|
292
|
-
*
|
|
293
|
-
*
|
|
294
|
-
* misstates why a defect shipped
|
|
295
|
-
* reads as an exhausted fixer when nothing was ever attempted.
|
|
288
|
+
* Four disjoint branches reach the terminal auto-ACCEPT, and only ONE of them has
|
|
289
|
+
* spent the autofix budget. An UNOBSERVED FAIL never consults the research; a
|
|
290
|
+
* frozen-blocked one is a contradiction no re-run can resolve; an ACCEPT
|
|
291
|
+
* recommendation can arrive with the budget untouched. A single "autofix budget
|
|
292
|
+
* spent" line would be false on three of the four, and a durable trail that
|
|
293
|
+
* misstates why a defect shipped reads as an exhausted fixer that never tried.
|
|
296
294
|
*/
|
|
297
295
|
export declare function yoloAcceptReason(c: YoloAcceptContext): string;
|
|
298
296
|
/**
|
|
@@ -306,8 +304,8 @@ export declare function askVerifyResolution(ctx: ExtensionCommandContext, title:
|
|
|
306
304
|
/**
|
|
307
305
|
* Run the verify + enforce gates against a task's just-finished implementation.
|
|
308
306
|
*
|
|
309
|
-
*
|
|
310
|
-
*
|
|
307
|
+
* Both commands gate identically because both call this. Returns a GateResult;
|
|
308
|
+
* `done` means the caller should proceed (the
|
|
311
309
|
* work is verified-or-accepted, checked off, committed, and enforced), every other
|
|
312
310
|
* kind is a terminal stop the caller announces. Never throws for a gate outcome —
|
|
313
311
|
* only a user cancel inside a gate child propagates (handled by the caller's
|
|
@@ -346,12 +344,10 @@ type VerifyGateStep = {
|
|
|
346
344
|
* recommendation → unattended autofix → picker) until it verifies, is accepted,
|
|
347
345
|
* or terminates.
|
|
348
346
|
*
|
|
349
|
-
* Split
|
|
350
|
-
*
|
|
351
|
-
*
|
|
352
|
-
*
|
|
353
|
-
* why `deps.verify` was driven by an invocation counter whose first return existed
|
|
354
|
-
* only to unlock `mode === 'edit'`.
|
|
347
|
+
* Split from `runGatesForTask` at the single boolean that crosses to the ENFORCE
|
|
348
|
+
* half (`cleanPass`). This loop has four terminal exits and carries the whole
|
|
349
|
+
* negotiation; enforce always falls through. Joined, a test of the enforce
|
|
350
|
+
* differential would have to traverse this entire loop first.
|
|
355
351
|
*/
|
|
356
352
|
export declare function resolveVerifyGate(ctxIn: ExtensionCommandContext, deps: GateDeps, p: GateParams, rec: Recorder, routeRootCause: RootCauseRouter): Promise<VerifyGateStep>;
|
|
357
353
|
/**
|
|
@@ -359,8 +355,7 @@ export declare function resolveVerifyGate(ctxIn: ExtensionCommandContext, deps:
|
|
|
359
355
|
* then decide whether the pass\'s own commit survives.
|
|
360
356
|
*
|
|
361
357
|
* `reVerify` is deliberately NOT `deps.verify`. They answer two different
|
|
362
|
-
* questions — the gate above, and this differential —
|
|
363
|
-
* field the only way to answer them differently was to count invocations.
|
|
358
|
+
* questions — the gate above, and this differential — so they are two fields.
|
|
364
359
|
*
|
|
365
360
|
* Reads `active` and never reassigns it: nothing here can replace the live
|
|
366
361
|
* session, unlike the autofix in the verify half.
|