@mjasnikovs/pi-task 0.38.29 → 0.38.31
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config/config.d.ts +70 -70
- package/dist/config/config.js +26 -35
- package/dist/config/extension-list.d.ts +6 -5
- package/dist/config/extension-list.js +3 -2
- package/dist/config/reasoning-args.d.ts +9 -7
- package/dist/config/reasoning-args.js +12 -10
- package/dist/config/reasoning.d.ts +44 -105
- package/dist/config/reasoning.js +27 -704
- package/dist/config/register.d.ts +34 -48
- package/dist/config/register.js +41 -51
- package/dist/config/tool-list.d.ts +16 -16
- package/dist/config/tool-list.js +1 -1
- package/dist/index.js +2 -0
- package/dist/remote/bridge.d.ts +19 -10
- package/dist/remote/bridge.js +3 -2
- package/dist/remote/broadcast.js +3 -1
- package/dist/remote/events.js +12 -11
- package/dist/remote/history.d.ts +1 -1
- package/dist/remote/protocol.d.ts +6 -3
- package/dist/remote/protocol.js +2 -1
- package/dist/remote/push.d.ts +16 -16
- package/dist/remote/push.js +27 -27
- package/dist/remote/register.d.ts +3 -3
- package/dist/remote/register.js +17 -19
- package/dist/remote/server.d.ts +9 -8
- package/dist/remote/server.js +15 -14
- package/dist/remote/session-state.d.ts +5 -4
- package/dist/remote/session-state.js +8 -5
- package/dist/remote/sw.d.ts +7 -6
- package/dist/remote/sw.js +7 -6
- package/dist/remote/tailscale.d.ts +4 -2
- package/dist/remote/tailscale.js +4 -2
- package/dist/remote/ui-highlight.js +6 -5
- package/dist/remote/ui-render.js +4 -4
- package/dist/remote/ui-script.js +24 -24
- package/dist/remote/ui-styles.d.ts +1 -1
- package/dist/remote/ui-styles.js +10 -13
- package/dist/remote/ui-tools.js +9 -6
- package/dist/shared/child-extensions.d.ts +29 -17
- package/dist/shared/child-extensions.js +29 -17
- package/dist/shared/child-output.d.ts +30 -24
- package/dist/shared/child-output.js +25 -17
- package/dist/shared/child-process.d.ts +47 -40
- package/dist/shared/child-process.js +50 -59
- package/dist/shared/command-watchdog.d.ts +85 -16
- package/dist/shared/command-watchdog.js +115 -21
- package/dist/shared/fs-text.d.ts +16 -10
- package/dist/shared/fs-text.js +16 -10
- package/dist/shared/git-runner.d.ts +25 -25
- package/dist/shared/git-runner.js +25 -25
- package/dist/shared/leaked-tool-call.d.ts +17 -11
- package/dist/shared/leaked-tool-call.js +23 -15
- package/dist/shared/model-endpoint.d.ts +29 -16
- package/dist/shared/model-endpoint.js +33 -21
- package/dist/shared/pi-invocation.d.ts +7 -4
- package/dist/shared/pi-invocation.js +12 -7
- package/dist/shared/pkg-version.d.ts +13 -5
- package/dist/shared/pkg-version.js +13 -5
- package/dist/shared/reasoning-capability.d.ts +35 -24
- package/dist/shared/reasoning-capability.js +35 -24
- package/dist/shared/stream-watchdog.d.ts +60 -44
- package/dist/shared/stream-watchdog.js +62 -45
- package/dist/task/accept-debt.d.ts +41 -43
- package/dist/task/accept-debt.js +73 -65
- package/dist/task/api-synthesis.d.ts +24 -21
- package/dist/task/api-synthesis.js +32 -26
- package/dist/task/apis-contract.d.ts +32 -64
- package/dist/task/apis-contract.js +32 -64
- package/dist/task/artifact-closure.d.ts +27 -13
- package/dist/task/artifact-closure.js +95 -67
- package/dist/task/auto-commit.d.ts +46 -35
- package/dist/task/auto-commit.js +51 -38
- package/dist/task/auto-io.d.ts +45 -25
- package/dist/task/auto-io.js +57 -29
- package/dist/task/auto-orchestrator.d.ts +26 -24
- package/dist/task/auto-orchestrator.js +192 -165
- package/dist/task/auto-prompts.d.ts +36 -24
- package/dist/task/auto-prompts.js +40 -26
- package/dist/task/autofix-ledger.d.ts +27 -25
- package/dist/task/autofix-ledger.js +29 -26
- package/dist/task/batch-test-task.d.ts +20 -12
- package/dist/task/batch-test-task.js +67 -60
- package/dist/task/boot-probe.d.ts +60 -44
- package/dist/task/boot-probe.js +91 -72
- package/dist/task/cancel-input.d.ts +30 -16
- package/dist/task/cancel-input.js +20 -11
- package/dist/task/cancel-points.d.ts +27 -20
- package/dist/task/cancel-points.js +30 -22
- package/dist/task/child-runner.d.ts +124 -55
- package/dist/task/child-runner.js +298 -90
- package/dist/task/child-status.d.ts +23 -16
- package/dist/task/child-status.js +23 -16
- package/dist/task/clamp-output.js +12 -5
- package/dist/task/command-run.d.ts +31 -28
- package/dist/task/command-run.js +44 -35
- package/dist/task/command-shrink.d.ts +25 -18
- package/dist/task/command-shrink.js +37 -31
- package/dist/task/command-watchdog.d.ts +9 -6
- package/dist/task/command-watchdog.js +21 -15
- package/dist/task/context-attribution.d.ts +34 -26
- package/dist/task/context-attribution.js +34 -26
- package/dist/task/context-silence.d.ts +39 -29
- package/dist/task/context-silence.js +35 -25
- package/dist/task/context-usage.d.ts +16 -9
- package/dist/task/context-usage.js +16 -9
- package/dist/task/contracts.d.ts +8 -4
- package/dist/task/contracts.js +25 -17
- package/dist/task/coverage-loop.d.ts +22 -18
- package/dist/task/coverage-loop.js +35 -30
- package/dist/task/critique-probes.d.ts +13 -14
- package/dist/task/critique-probes.js +50 -39
- package/dist/task/debug-log.d.ts +13 -5
- package/dist/task/debug-log.js +32 -20
- package/dist/task/decompose-fidelity.d.ts +11 -9
- package/dist/task/decompose-fidelity.js +38 -33
- package/dist/task/decompose-granularity.d.ts +41 -38
- package/dist/task/decompose-granularity.js +41 -38
- package/dist/task/deep-render-check.d.ts +22 -14
- package/dist/task/deep-render-check.js +40 -31
- package/dist/task/dropped-input.d.ts +12 -7
- package/dist/task/dropped-input.js +5 -2
- package/dist/task/enforce-attribution.d.ts +38 -47
- package/dist/task/enforce-attribution.js +46 -52
- package/dist/task/enforce-guidelines.d.ts +31 -20
- package/dist/task/enforce-guidelines.js +32 -21
- package/dist/task/enrichment.d.ts +7 -2
- package/dist/task/enrichment.js +26 -14
- package/dist/task/env-notes.d.ts +16 -7
- package/dist/task/env-notes.js +48 -31
- package/dist/task/env-template-closure.d.ts +4 -4
- package/dist/task/env-template-closure.js +42 -34
- package/dist/task/external-context.d.ts +28 -21
- package/dist/task/external-context.js +17 -12
- package/dist/task/failure-classifier.d.ts +4 -5
- package/dist/task/failure-classifier.js +30 -8
- package/dist/task/file-inventory.d.ts +15 -11
- package/dist/task/file-inventory.js +25 -22
- package/dist/task/final-gate-fix.d.ts +74 -86
- package/dist/task/final-gate-fix.js +97 -116
- package/dist/task/final-gate-progress.d.ts +29 -46
- package/dist/task/final-gate-progress.js +40 -51
- package/dist/task/final-gate.d.ts +64 -97
- package/dist/task/final-gate.js +192 -199
- package/dist/task/fix-child.d.ts +21 -27
- package/dist/task/fix-child.js +21 -27
- package/dist/task/foreign-path.d.ts +6 -5
- package/dist/task/foreign-path.js +0 -0
- package/dist/task/frozen-conflict.d.ts +9 -10
- package/dist/task/frozen-conflict.js +61 -64
- package/dist/task/frozen-path-guard.d.ts +35 -14
- package/dist/task/frozen-path-guard.js +56 -39
- package/dist/task/gate-child.d.ts +27 -28
- package/dist/task/gate-child.js +36 -35
- package/dist/task/gate-deps.d.ts +34 -27
- package/dist/task/gate-deps.js +169 -159
- package/dist/task/gate-tally.d.ts +77 -80
- package/dist/task/gate-tally.js +65 -68
- package/dist/task/git-state-guard.d.ts +15 -11
- package/dist/task/git-state-guard.js +76 -66
- package/dist/task/impl-widget.d.ts +25 -16
- package/dist/task/impl-widget.js +27 -17
- package/dist/task/implementation-guards.d.ts +26 -0
- package/dist/task/implementation-guards.js +177 -0
- package/dist/task/implementation-thinking.d.ts +33 -31
- package/dist/task/implementation-thinking.js +5 -6
- package/dist/task/implementation-turn.d.ts +39 -31
- package/dist/task/implementation-turn.js +41 -28
- package/dist/task/inline-markdown.d.ts +20 -7
- package/dist/task/inline-markdown.js +15 -6
- package/dist/task/launch-config-gap.js +25 -39
- package/dist/task/launch-contract.d.ts +18 -21
- package/dist/task/launch-contract.js +28 -30
- package/dist/task/launch-manifest.d.ts +6 -2
- package/dist/task/launch-manifest.js +35 -34
- package/dist/task/ledger.js +16 -14
- package/dist/task/lint-fix.d.ts +6 -8
- package/dist/task/lint-fix.js +67 -69
- package/dist/task/loop-detector.d.ts +27 -8
- package/dist/task/loop-detector.js +38 -14
- package/dist/task/mid-run-input.d.ts +17 -15
- package/dist/task/mid-run-input.js +17 -15
- package/dist/task/orchestrator.d.ts +24 -28
- package/dist/task/orchestrator.js +89 -66
- package/dist/task/orientation.d.ts +18 -23
- package/dist/task/orientation.js +24 -31
- package/dist/task/owned-freeze-conflict.d.ts +21 -20
- package/dist/task/owned-freeze-conflict.js +52 -85
- package/dist/task/owned-freeze-reassign.d.ts +40 -60
- package/dist/task/owned-freeze-reassign.js +41 -61
- package/dist/task/parsers.d.ts +4 -2
- package/dist/task/parsers.js +4 -4
- package/dist/task/phases.d.ts +41 -48
- package/dist/task/phases.js +196 -252
- package/dist/task/plan-io.d.ts +6 -7
- package/dist/task/plan-io.js +6 -7
- package/dist/task/plan-orchestrator.d.ts +10 -8
- package/dist/task/plan-orchestrator.js +14 -10
- package/dist/task/plan-prompts.d.ts +6 -5
- package/dist/task/plan-prompts.js +6 -5
- package/dist/task/plan-readonly.d.ts +4 -5
- package/dist/task/plan-readonly.js +4 -5
- package/dist/task/plan-rounds.d.ts +17 -29
- package/dist/task/plan-rounds.js +21 -34
- package/dist/task/plan-session.d.ts +58 -72
- package/dist/task/plan-session.js +61 -83
- package/dist/task/probe-gaming.d.ts +28 -27
- package/dist/task/probe-gaming.js +0 -0
- package/dist/task/prohibition-probe.d.ts +14 -16
- package/dist/task/prompts.d.ts +3 -4
- package/dist/task/prompts.js +17 -26
- package/dist/task/qa-transcript.d.ts +15 -22
- package/dist/task/qa-transcript.js +15 -21
- package/dist/task/question-box.d.ts +17 -13
- package/dist/task/question-box.js +19 -15
- package/dist/task/question-dedup.d.ts +6 -7
- package/dist/task/question-dedup.js +13 -14
- package/dist/task/question-dialog.d.ts +22 -32
- package/dist/task/question-dialog.js +22 -32
- package/dist/task/question-source.d.ts +18 -44
- package/dist/task/question-source.js +22 -51
- package/dist/task/refuted-constraint.d.ts +11 -31
- package/dist/task/refuted-constraint.js +27 -51
- package/dist/task/regenerable-artifacts.d.ts +12 -31
- package/dist/task/regenerable-artifacts.js +12 -31
- package/dist/task/render-check.d.ts +11 -22
- package/dist/task/render-check.js +33 -46
- package/dist/task/repo-health-check.d.ts +10 -14
- package/dist/task/repo-health-check.js +17 -23
- package/dist/task/requirements.d.ts +38 -71
- package/dist/task/requirements.js +78 -126
- package/dist/task/research-fanout-budget.d.ts +51 -88
- package/dist/task/research-fanout-budget.js +51 -88
- package/dist/task/research-worker.d.ts +29 -39
- package/dist/task/research-worker.js +37 -61
- package/dist/task/resume-gap.d.ts +14 -15
- package/dist/task/root-cause-repair.d.ts +9 -9
- package/dist/task/root-cause-repair.js +28 -40
- package/dist/task/run-bracket.d.ts +10 -13
- package/dist/task/run-end.d.ts +12 -22
- package/dist/task/run-end.js +8 -16
- package/dist/task/run-final-gate.d.ts +19 -21
- package/dist/task/run-final-gate.js +62 -80
- package/dist/task/runner-globs.d.ts +12 -13
- package/dist/task/runner-globs.js +12 -13
- package/dist/task/runner-resolve.d.ts +9 -9
- package/dist/task/runner-resolve.js +22 -23
- package/dist/task/script-escape.d.ts +10 -12
- package/dist/task/script-escape.js +13 -14
- package/dist/task/serve-entry.d.ts +1 -1
- package/dist/task/serve-entry.js +22 -25
- package/dist/task/service-blocks.js +4 -2
- package/dist/task/shipped-source.d.ts +11 -29
- package/dist/task/shipped-source.js +11 -29
- package/dist/task/skip-escape.js +10 -14
- package/dist/task/spec-urls.d.ts +26 -65
- package/dist/task/spec-urls.js +26 -65
- package/dist/task/spec-validation.d.ts +17 -20
- package/dist/task/spec-validation.js +17 -20
- package/dist/task/stall-detector.d.ts +23 -30
- package/dist/task/stall-detector.js +23 -30
- package/dist/task/stream-watchdog.d.ts +14 -12
- package/dist/task/stream-watchdog.js +14 -12
- package/dist/task/substitution-probe.d.ts +17 -20
- package/dist/task/substitution-probe.js +17 -20
- package/dist/task/task-gates.d.ts +36 -41
- package/dist/task/task-gates.js +95 -106
- package/dist/task/task-io.d.ts +4 -4
- package/dist/task/task-io.js +4 -4
- package/dist/task/task-parsers.js +4 -3
- package/dist/task/task-provenance.d.ts +2 -2
- package/dist/task/task-provenance.js +11 -13
- package/dist/task/task-types.d.ts +4 -3
- package/dist/task/terminal-outcome.d.ts +14 -16
- package/dist/task/terminal-outcome.js +12 -14
- package/dist/task/test-assembly.d.ts +13 -20
- package/dist/task/test-assembly.js +13 -20
- package/dist/task/timings.d.ts +5 -3
- package/dist/task/timings.js +5 -3
- package/dist/task/title-label.d.ts +9 -4
- package/dist/task/title-label.js +9 -4
- package/dist/task/type-only-answer.d.ts +44 -52
- package/dist/task/type-only-answer.js +44 -52
- package/dist/task/unfailable-command.d.ts +18 -24
- package/dist/task/unfailable-command.js +21 -27
- package/dist/task/unknown-routing.d.ts +10 -4
- package/dist/task/unknown-routing.js +10 -4
- package/dist/task/user-directives.d.ts +5 -8
- package/dist/task/user-directives.js +5 -8
- package/dist/task/verify-quality.d.ts +18 -22
- package/dist/task/verify-quality.js +45 -46
- package/dist/task/verify-reconcile.d.ts +15 -10
- package/dist/task/verify-reconcile.js +45 -43
- package/dist/task/verify-resolution.d.ts +24 -20
- package/dist/task/verify-resolution.js +51 -50
- package/dist/task/verify-work.d.ts +59 -66
- package/dist/task/verify-work.js +101 -138
- package/dist/task/widget.d.ts +15 -14
- package/dist/task/widget.js +22 -17
- package/dist/task/wiring-claims.d.ts +25 -32
- package/dist/task/wiring-claims.js +30 -35
- package/dist/task/write-guard.d.ts +39 -39
- package/dist/task/write-guard.js +48 -51
- package/dist/task/yolo.d.ts +34 -30
- package/dist/task/yolo.js +42 -37
- package/dist/workers/abstention.d.ts +21 -41
- package/dist/workers/abstention.js +27 -48
- package/dist/workers/brave-search.d.ts +4 -3
- package/dist/workers/brave-search.js +5 -2
- package/dist/workers/brave-warning.d.ts +7 -4
- package/dist/workers/brave-warning.js +19 -7
- package/dist/workers/ddg-search.d.ts +6 -6
- package/dist/workers/ddg-search.js +18 -12
- package/dist/workers/docs-cache.js +5 -2
- package/dist/workers/docs-chunk.d.ts +30 -37
- package/dist/workers/docs-chunk.js +37 -41
- package/dist/workers/docs-core.d.ts +28 -44
- package/dist/workers/docs-core.js +25 -44
- package/dist/workers/docs-index.js +4 -3
- package/dist/workers/docs-lookup.d.ts +15 -22
- package/dist/workers/docs-lookup.js +12 -21
- package/dist/workers/docs-project.d.ts +15 -9
- package/dist/workers/docs-project.js +17 -10
- package/dist/workers/docs-resolve.d.ts +19 -20
- package/dist/workers/docs-resolve.js +35 -32
- package/dist/workers/docs-retrieve.d.ts +5 -6
- package/dist/workers/docs-retrieve.js +18 -15
- package/dist/workers/exa-search.d.ts +9 -6
- package/dist/workers/exa-search.js +23 -12
- package/dist/workers/fetch-core.d.ts +13 -16
- package/dist/workers/fetch-core.js +23 -23
- package/dist/workers/focused-extractor.d.ts +13 -12
- package/dist/workers/focused-extractor.js +27 -19
- package/dist/workers/html-clean.js +24 -14
- package/dist/workers/http-request.d.ts +28 -20
- package/dist/workers/http-request.js +22 -17
- package/dist/workers/npm-version.d.ts +28 -11
- package/dist/workers/npm-version.js +24 -15
- package/dist/workers/phantom-imports.d.ts +15 -12
- package/dist/workers/phantom-imports.js +30 -24
- package/dist/workers/pi-worker-core.d.ts +65 -96
- package/dist/workers/pi-worker-core.js +93 -181
- package/dist/workers/pi-worker-docs.d.ts +24 -19
- package/dist/workers/pi-worker-docs.js +67 -76
- package/dist/workers/pi-worker-fetch.d.ts +7 -3
- package/dist/workers/pi-worker-fetch.js +27 -19
- package/dist/workers/pi-worker-search.js +12 -8
- package/dist/workers/pi-worker.d.ts +9 -4
- package/dist/workers/pi-worker.js +21 -14
- package/dist/workers/reasoning-warning.d.ts +18 -17
- package/dist/workers/reasoning-warning.js +22 -20
- package/dist/workers/research-cache.js +50 -78
- package/dist/workers/search-core.js +7 -5
- package/dist/workers/search-types.d.ts +10 -9
- package/dist/workers/search-types.js +9 -8
- package/dist/workers/session-hint.d.ts +13 -14
- package/dist/workers/session-hint.js +8 -9
- package/dist/workers/shared.d.ts +21 -25
- package/dist/workers/shared.js +0 -0
- package/dist/workers/single-read-extension.d.ts +14 -7
- package/dist/workers/single-read-extension.js +14 -7
- package/dist/workers/single-read-guard.d.ts +27 -30
- package/dist/workers/single-read-guard.js +36 -36
- package/dist/workers/typeonly-log.d.ts +12 -9
- package/dist/workers/typeonly-log.js +29 -33
- package/dist/workers/worker-channels.d.ts +15 -23
- package/dist/workers/worker-channels.js +15 -23
- package/dist/workers/worker-failure.d.ts +38 -46
- package/dist/workers/worker-failure.js +31 -39
- package/dist/workers/worker-kill.d.ts +25 -26
- package/dist/workers/worker-kill.js +16 -19
- package/dist/workers/worker-profiles.d.ts +54 -56
- package/dist/workers/worker-profiles.js +63 -39
- package/package.json +10 -8
|
@@ -1,74 +1,48 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
* THE FAULT THEY TARGET (mx5 run 18, measured — scripts/research-restart-baserate.ts):
|
|
18
|
-
* `worker:apis` fans out `pi-worker-docs(module: ".")` project-source lookups, each
|
|
19
|
-
* of which spawns its own summarising child, and the per-worker wall-clock cap is
|
|
20
|
-
* 240s. Pearson r(project lookups, worker wall clock) = 0.909 over 24 tasks. 0-4
|
|
21
|
-
* lookups never timed out; every worker at >=46 lookups burned the FULL restart
|
|
22
|
-
* budget — 3 attempts, 720s, two of them discarded whole. The 240s ceiling and a
|
|
23
|
-
* 46-call fan-out are jointly unsatisfiable, so the timeout is not a backstop
|
|
24
|
-
* there, it is the guaranteed outcome.
|
|
25
|
-
*
|
|
26
|
-
* TWO WAYS TO MAKE THEM SATISFIABLE, and the A/B — not this comment — decides:
|
|
2
|
+
* research-fanout-budget — the levers bounding worker:apis's project-source
|
|
3
|
+
* fan-out, and which of them is on.
|
|
4
|
+
*
|
|
5
|
+
* `workerProgressCeilingMs` is ON by default; its env var is the OFF switch. CAP
|
|
6
|
+
* (`projectDocsBudget`), SCALE (`fanoutTimeoutPolicy`) and RESCUE-CARRY
|
|
7
|
+
* (`workerCarryForward`) are OFF unless their env var is set. They stay in the
|
|
8
|
+
* shipped build so a harness can run them against the shipped baseline in the SAME
|
|
9
|
+
* build — patching a copy of the code measures the copy, not the code. Nothing may
|
|
10
|
+
* read them outside such a harness.
|
|
11
|
+
*
|
|
12
|
+
* THE FAULT THEY TARGET. `worker:apis` fans out `pi-worker-docs(module: ".")`
|
|
13
|
+
* project-source lookups, and each one spawns its own summarising child. Under a
|
|
14
|
+
* fixed wall-clock cap, a large enough fan-out cannot finish inside it — so the
|
|
15
|
+
* timeout is not a backstop there, it is the guaranteed outcome.
|
|
27
16
|
*
|
|
28
17
|
* CAP bound the fan-out to fit the ceiling. Told to the worker upfront
|
|
29
18
|
* (projectDocsBudgetNotice) and enforced in the tool
|
|
30
|
-
* (projectDocsBudgetExhausted)
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
* SCALE bound the ceiling to fit the fan-out: each project-source lookup
|
|
34
|
-
*
|
|
35
|
-
*
|
|
36
|
-
*
|
|
37
|
-
*
|
|
38
|
-
*
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
* spend the extra time and still time out, buying nothing.
|
|
42
|
-
*
|
|
43
|
-
* ─────────────────────────────────────────────────────────────────────────────
|
|
44
|
-
* BOTH OF THE ABOVE ANSWER THE WRONG QUESTION. Kept for the record and for the
|
|
45
|
-
* A/B's other arms, but they are not the fix.
|
|
46
|
-
*
|
|
47
|
-
* They argue about how long a worker may run. The actual defect is what happens
|
|
48
|
-
* when it runs out: the attempt is killed and everything it produced is THROWN
|
|
49
|
-
* AWAY, and the re-spawn is given a hint but no findings — so it re-reads the
|
|
50
|
-
* same files against the same clock and dies in the same place. That is why
|
|
51
|
-
* every worker at >=46 lookups burned the FULL budget rather than converging.
|
|
52
|
-
* The r=0.909 correlation measures the amnesia, not an over-long task.
|
|
19
|
+
* (projectDocsBudgetExhausted). The prompt alone does not bind: the same
|
|
20
|
+
* worker is ALREADY told "be decisive" by WORKER_TIMEOUT_HINT
|
|
21
|
+
* (pi-worker-core.ts) on every restart.
|
|
22
|
+
* SCALE bound the ceiling to fit the fan-out: each project-source lookup pushes
|
|
23
|
+
* the deadline out, up to a hard ceiling, so a worker that is making
|
|
24
|
+
* progress is not killed for making progress.
|
|
25
|
+
*
|
|
26
|
+
* BOTH ANSWER THE WRONG QUESTION. They argue about how long a worker may run. The
|
|
27
|
+
* defect is what happens when it runs out: the attempt is killed, everything it
|
|
28
|
+
* produced is THROWN AWAY, and the re-spawn gets a hint but no findings — so it
|
|
29
|
+
* re-reads the same files against the same clock and dies in the same place.
|
|
53
30
|
*
|
|
54
31
|
* Judged against "the worker must return its work", CAP makes the worker read
|
|
55
|
-
* LESS
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
*
|
|
60
|
-
*
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
*
|
|
67
|
-
*
|
|
68
|
-
*
|
|
69
|
-
* entry replayed under "work already done" is exactly how a fabrication gets
|
|
70
|
-
* laundered into a final answer. Hence the carry is framed as unverified, and
|
|
71
|
-
* ungrounded-symbol and anti-synthesis counts gate the arm.
|
|
32
|
+
* LESS, lowering the requirement so the metric goes green; and SCALE is a per-file
|
|
33
|
+
* constant that dies on one big file and, being wall-clock, makes answer quality a
|
|
34
|
+
* function of the user's hardware — the same task on a slower model loses its work.
|
|
35
|
+
*
|
|
36
|
+
* RESCUE (pi-worker-core.ts) carry the killed attempt's findings into the next
|
|
37
|
+
* one and never return less than the best attempt produced, so a restart
|
|
38
|
+
* CONVERGES instead of repeating; and deadline on lack of PROGRESS rather
|
|
39
|
+
* than elapsed time, so "slow" and "stuck" stop being the same verdict.
|
|
40
|
+
* Being stuck is already detected separately by the output-stall probe
|
|
41
|
+
* (STALL_AFTER_MS, worker-profiles.ts), which resets on progress.
|
|
42
|
+
*
|
|
43
|
+
* The carry has a risk of its own: a half-written entry replayed under "work
|
|
44
|
+
* already done" is how a fabrication gets laundered into a final answer. That is
|
|
45
|
+
* why the carry is framed to the worker as unverified.
|
|
72
46
|
*/
|
|
73
47
|
/** Max project-source (`module: "."`) docs lookups per worker ATTEMPT. Unset = no cap. */
|
|
74
48
|
export const PROJECT_DOCS_BUDGET_ENV = 'PI_TASK_PROJECT_DOCS_BUDGET';
|
|
@@ -99,12 +73,10 @@ export const RESEARCH_LEVER_ENVS = [
|
|
|
99
73
|
* The levers, read ONCE, as a reader the profile table can be handed.
|
|
100
74
|
*
|
|
101
75
|
* WHY A SNAPSHOT AND NOT `process.env`. Every worker in one research phase must
|
|
102
|
-
* see the same
|
|
103
|
-
*
|
|
104
|
-
*
|
|
105
|
-
*
|
|
106
|
-
* var mid-phase would then half-apply its own arm. Freezing the reader keeps the
|
|
107
|
-
* read-once property while letting the profile own what the values MEAN.
|
|
76
|
+
* see the same lever values. A profile that read `process.env` itself would move
|
|
77
|
+
* the read down to each worker, and a var flipped mid-phase would then apply to
|
|
78
|
+
* some workers and not others. Freezing the reader keeps the read-once property
|
|
79
|
+
* while letting the profile own what the values MEAN.
|
|
108
80
|
*/
|
|
109
81
|
export function snapshotLeverEnv(env = defaultEnv) {
|
|
110
82
|
const snap = new Map(RESEARCH_LEVER_ENVS.map(k => [k, env(k)]));
|
|
@@ -149,14 +121,10 @@ export function workerCarryForward(env = defaultEnv) {
|
|
|
149
121
|
/**
|
|
150
122
|
* The absolute backstop for the progress-based deadline.
|
|
151
123
|
*
|
|
152
|
-
*
|
|
153
|
-
*
|
|
154
|
-
*
|
|
155
|
-
*
|
|
156
|
-
* sit clear of the real workload. Measured on 42 progress-arm trials
|
|
157
|
-
* (`~/tmp/research-fanout-ab-v3`): median 275s, p90 523s, **max 730s**. 20 minutes
|
|
158
|
-
* is 1.6x the observed worst case, and 1.7x the 720s the SHIPPED path already
|
|
159
|
-
* spends on a worker that burns all three attempts and returns nothing.
|
|
124
|
+
* It is not a budget and it does not decide how long a worker may take — the
|
|
125
|
+
* no-progress deadline does that, and it resets on every tool call. This is the
|
|
126
|
+
* last-resort bound on a worker that never stops moving (a tool-call loop the loop
|
|
127
|
+
* detector misses), so its only requirement is to sit clear of the real workload.
|
|
160
128
|
*
|
|
161
129
|
* A ceiling that never fires in production is the correct behaviour for a
|
|
162
130
|
* backstop, not evidence it is untested: it fires under test
|
|
@@ -168,12 +136,7 @@ export const DEFAULT_WORKER_PROGRESS_CEILING_MS = 1_200_000;
|
|
|
168
136
|
/**
|
|
169
137
|
* The progress-based deadline's ceiling, or null when the lever is OFF.
|
|
170
138
|
*
|
|
171
|
-
*
|
|
172
|
-
* switch. Measured baseline vs progress over 42 trials/arm on a calibrated
|
|
173
|
-
* instrument (A/A false-break 1.5%): worker-timeout restarts 22/24 → 0/24,
|
|
174
|
-
* degrades 8/24 → 0/24, entries up on all four high-fan-out fixtures (TASK_0021
|
|
175
|
-
* 11.0 → 25.5), quality invariants HOLD, every treatment-arm ungrounded flag
|
|
176
|
-
* hand-verified as an instrument artifact rather than a fabrication.
|
|
139
|
+
* ON by default, so the env var is the OFF switch, not the on switch:
|
|
177
140
|
*
|
|
178
141
|
* unset ON at DEFAULT_WORKER_PROGRESS_CEILING_MS
|
|
179
142
|
* "0" | "off" OFF — the fixed elapsed-time cap, exactly as before
|
|
@@ -194,9 +157,9 @@ export function workerProgressCeilingMs(env = defaultEnv) {
|
|
|
194
157
|
* The upfront half of the CAP arm, appended to the APIS worker's prompt.
|
|
195
158
|
*
|
|
196
159
|
* Upfront and NUMERIC on purpose. The worker cannot ration a budget it learns
|
|
197
|
-
* about only when it is spent, and "be decisive" — which it
|
|
198
|
-
* every timeout restart — is
|
|
199
|
-
*
|
|
160
|
+
* about only when it is spent, and "be decisive" — WORKER_TIMEOUT_HINT, which it
|
|
161
|
+
* already receives on every timeout restart — is the unquantified version of the
|
|
162
|
+
* same ask.
|
|
200
163
|
*/
|
|
201
164
|
export function projectDocsBudgetNotice(budget) {
|
|
202
165
|
return (`\n\nLOOKUP BUDGET: you may make at most ${budget} project-source `
|
|
@@ -1,21 +1,17 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* ONE research worker, cache-skip to persist.
|
|
3
3
|
*
|
|
4
|
-
* WHY IT IS A MODULE.
|
|
5
|
-
*
|
|
6
|
-
* the
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
* legitimately empty section or ships one written from memory.
|
|
4
|
+
* WHY IT IS A MODULE. It holds the three RETRY GATES and their precedence: the
|
|
5
|
+
* EMPTY-SECTION gate (the only one that can fail the phase), the ZERO-RETRIEVAL
|
|
6
|
+
* gate and the SILENT gate (both of which discard a failed retry and ship the
|
|
7
|
+
* original). Getting that order wrong is how a run either dies on a legitimately
|
|
8
|
+
* empty section or ships one written from memory.
|
|
10
9
|
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* tests from testing the gate. Behind this interface a test scripts
|
|
17
|
-
* `runWorker(label, attempt)` and states the `RunWorkerResult` fields a gate
|
|
18
|
-
* reads.
|
|
10
|
+
* Behind `ResearchWorkerRun` a test scripts `runWorker(label, attempt)` and states
|
|
11
|
+
* the `RunWorkerResult` fields a gate reads. Reaching a gate through the phase
|
|
12
|
+
* instead would mean a temp dir, a real task file, and a fake spawn routed on
|
|
13
|
+
* prose lifted out of `prompts.ts` — so re-wording a prompt would silently stop
|
|
14
|
+
* the gate tests from testing the gate.
|
|
19
15
|
*
|
|
20
16
|
* THE INVARIANT, stated once: the empty gate runs FIRST and is the only one that
|
|
21
17
|
* can throw; the other two keep the original when their retry does not improve a
|
|
@@ -27,8 +23,8 @@ import type { SpawnFn } from '../shared/child-process.js';
|
|
|
27
23
|
import type { DebugLine } from './debug-log.js';
|
|
28
24
|
/**
|
|
29
25
|
* One research worker's row. `section` is the heading its output is assembled
|
|
30
|
-
* and cached under; `label` is its child NAME — what the loader
|
|
31
|
-
*
|
|
26
|
+
* and cached under; `label` is its child NAME — what the loader and the debug
|
|
27
|
+
* trail print, and the key into `REASONING_GROUP_BY_CHILD` (config/reasoning.ts).
|
|
32
28
|
*/
|
|
33
29
|
export interface ResearchWorkerSpec {
|
|
34
30
|
section: string;
|
|
@@ -52,7 +48,7 @@ export interface ResearchWorkerSpec {
|
|
|
52
48
|
* loop-degrade banner or a hallucinated non-bullet fragment — is re-run ONCE
|
|
53
49
|
* with this preamble. A legitimately-empty section is NOT retried. */
|
|
54
50
|
retryIfSilent?: string;
|
|
55
|
-
/** This worker can issue project-source docs lookups, so the
|
|
51
|
+
/** This worker can issue project-source docs lookups, so the fan-out bounds
|
|
56
52
|
* apply to it (see task/research-fanout-budget.ts). */
|
|
57
53
|
fanoutBounded?: true;
|
|
58
54
|
}
|
|
@@ -67,9 +63,8 @@ export interface ResearchWorkerRun {
|
|
|
67
63
|
runWorker: (label: string, input: RunWorkerInput) => Promise<RunWorkerResult>;
|
|
68
64
|
/**
|
|
69
65
|
* The parent session's context window, forwarded to every worker child.
|
|
70
|
-
*
|
|
71
|
-
* the churn rule
|
|
72
|
-
* `RunWorkerInput.contextWindow`.
|
|
66
|
+
* Nothing in pi's stream reports one, so a child that is not TOLD sits at 0 and
|
|
67
|
+
* the churn rule cannot fire — see `RunWorkerInput.contextWindow`.
|
|
73
68
|
*/
|
|
74
69
|
contextWindow: number | 'unknown';
|
|
75
70
|
cwd: string;
|
|
@@ -87,22 +82,20 @@ export interface ResearchWorkerRun {
|
|
|
87
82
|
/**
|
|
88
83
|
* Read one worker's cached output back, or '' when there is none.
|
|
89
84
|
*
|
|
90
|
-
* A SEAM, and the symmetric half of `persistSection`. The cache skip is one
|
|
91
|
-
*
|
|
92
|
-
* real task file on disk
|
|
93
|
-
* dir at all.
|
|
85
|
+
* A SEAM, and the symmetric half of `persistSection`. The cache skip is one of
|
|
86
|
+
* the four outcomes this driver has, and reaching it any other way would need a
|
|
87
|
+
* real task file on disk.
|
|
94
88
|
*/
|
|
95
89
|
readCached: (heading: string) => Promise<string>;
|
|
96
90
|
/** Write one validated section to the task file. Serialised by the caller. */
|
|
97
91
|
persistSection: (heading: string, text: string) => Promise<void>;
|
|
98
92
|
/**
|
|
99
|
-
* The
|
|
93
|
+
* The fan-out lever env vars, READ ONCE for the whole phase.
|
|
100
94
|
*
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
-
*
|
|
104
|
-
*
|
|
105
|
-
* per worker would lose. See `snapshotLeverEnv`.
|
|
95
|
+
* A frozen reader rather than resolved values, because the `research` profile
|
|
96
|
+
* (workers/worker-profiles.ts) owns what those values MEAN. What this layer
|
|
97
|
+
* owns is that every worker in one run sees the same values, which a live
|
|
98
|
+
* `process.env` read per worker would lose. See `snapshotLeverEnv`.
|
|
106
99
|
*/
|
|
107
100
|
leverEnv: (key: string) => string | undefined;
|
|
108
101
|
}
|
|
@@ -137,15 +130,12 @@ export declare function researchWorkerCacheHeading(section: string): string;
|
|
|
137
130
|
* wrote nothing): NOT a failure. On an extremely simple task ("create a folder
|
|
138
131
|
* with an index.html in it") three of the four workers have genuinely nothing
|
|
139
132
|
* to report, and each worker prompt tells the model to emit ONLY what this task
|
|
140
|
-
* touches and to drop everything else — so silence is the CORRECT answer and
|
|
141
|
-
*
|
|
142
|
-
*
|
|
143
|
-
*
|
|
144
|
-
*
|
|
145
|
-
*
|
|
146
|
-
* caller retries once and then accepts an explicit empty section; what stays
|
|
147
|
-
* fatal is silence WITH a reported cause, which is the masked-disconnect case
|
|
148
|
-
* this branch was written for and which `modelError` now names outright.
|
|
133
|
+
* touches and to drop everything else — so silence is the CORRECT answer, and
|
|
134
|
+
* treating it as a failure kills the whole task at research. "Nothing here"
|
|
135
|
+
* and zero bytes are the same answer; which one a model writes is style, not
|
|
136
|
+
* signal. The caller retries once and then accepts an explicit empty section.
|
|
137
|
+
* What stays fatal is silence WITH a reported cause, the masked-disconnect
|
|
138
|
+
* case `modelError` names outright.
|
|
149
139
|
*
|
|
150
140
|
* Returns null when the result is trustworthy.
|
|
151
141
|
*/
|
|
@@ -1,21 +1,17 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* ONE research worker, cache-skip to persist.
|
|
3
3
|
*
|
|
4
|
-
* WHY IT IS A MODULE.
|
|
5
|
-
*
|
|
6
|
-
* the
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
* legitimately empty section or ships one written from memory.
|
|
4
|
+
* WHY IT IS A MODULE. It holds the three RETRY GATES and their precedence: the
|
|
5
|
+
* EMPTY-SECTION gate (the only one that can fail the phase), the ZERO-RETRIEVAL
|
|
6
|
+
* gate and the SILENT gate (both of which discard a failed retry and ship the
|
|
7
|
+
* original). Getting that order wrong is how a run either dies on a legitimately
|
|
8
|
+
* empty section or ships one written from memory.
|
|
10
9
|
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
* tests from testing the gate. Behind this interface a test scripts
|
|
17
|
-
* `runWorker(label, attempt)` and states the `RunWorkerResult` fields a gate
|
|
18
|
-
* reads.
|
|
10
|
+
* Behind `ResearchWorkerRun` a test scripts `runWorker(label, attempt)` and states
|
|
11
|
+
* the `RunWorkerResult` fields a gate reads. Reaching a gate through the phase
|
|
12
|
+
* instead would mean a temp dir, a real task file, and a fake spawn routed on
|
|
13
|
+
* prose lifted out of `prompts.ts` — so re-wording a prompt would silently stop
|
|
14
|
+
* the gate tests from testing the gate.
|
|
19
15
|
*
|
|
20
16
|
* THE INVARIANT, stated once: the empty gate runs FIRST and is the only one that
|
|
21
17
|
* can throw; the other two keep the original when their retry does not improve a
|
|
@@ -57,15 +53,12 @@ export function researchWorkerCacheHeading(section) {
|
|
|
57
53
|
* wrote nothing): NOT a failure. On an extremely simple task ("create a folder
|
|
58
54
|
* with an index.html in it") three of the four workers have genuinely nothing
|
|
59
55
|
* to report, and each worker prompt tells the model to emit ONLY what this task
|
|
60
|
-
* touches and to drop everything else — so silence is the CORRECT answer and
|
|
61
|
-
*
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
* caller retries once and then accepts an explicit empty section; what stays
|
|
67
|
-
* fatal is silence WITH a reported cause, which is the masked-disconnect case
|
|
68
|
-
* this branch was written for and which `modelError` now names outright.
|
|
56
|
+
* touches and to drop everything else — so silence is the CORRECT answer, and
|
|
57
|
+
* treating it as a failure kills the whole task at research. "Nothing here"
|
|
58
|
+
* and zero bytes are the same answer; which one a model writes is style, not
|
|
59
|
+
* signal. The caller retries once and then accepts an explicit empty section.
|
|
60
|
+
* What stays fatal is silence WITH a reported cause, the masked-disconnect
|
|
61
|
+
* case `modelError` names outright.
|
|
69
62
|
*
|
|
70
63
|
* Returns null when the result is trustworthy.
|
|
71
64
|
*/
|
|
@@ -125,7 +118,7 @@ export function classifyResearchWorker(name, result) {
|
|
|
125
118
|
// point of this branch is to tell them apart:
|
|
126
119
|
//
|
|
127
120
|
// FAILED, cause reported: pi delivers a failed turn as an empty assistant
|
|
128
|
-
// message with stopReason "error" and exit 0, so the real cause
|
|
121
|
+
// message with stopReason "error" and exit 0, so the real cause would be
|
|
129
122
|
// discarded and reported as the useless "produced no output". Name it.
|
|
130
123
|
// FAILED, child never spoke: no stdout at all means the child died before it
|
|
131
124
|
// could run (unresolvable provider, missing key, bad argv) — it never
|
|
@@ -181,10 +174,9 @@ export function emptySectionBody(name) {
|
|
|
181
174
|
}
|
|
182
175
|
/**
|
|
183
176
|
* A worker answer that IS the word "nothing" and carries no other content:
|
|
184
|
-
* `(none)`, `N/A`, `- none`, `(no content)`, `(no entries)`.
|
|
185
|
-
* these
|
|
186
|
-
*
|
|
187
|
-
* and each one means exactly what an empty answer means — so they are recorded
|
|
177
|
+
* `(none)`, `N/A`, `- none`, `(no content)`, `(no entries)`. A worker writes one of
|
|
178
|
+
* these on a task that touches nothing, and each means exactly what an empty
|
|
179
|
+
* answer means — so they are recorded
|
|
188
180
|
* with the same marker rather than passed through in whatever shape the model
|
|
189
181
|
* happened to pick. Deliberately NARROW: it matches only a lone token, never
|
|
190
182
|
* prose like "(no APIs to list — this task creates a plain HTML file …)", which
|
|
@@ -205,16 +197,9 @@ export function isBareNoneAnswer(text) {
|
|
|
205
197
|
* fires on a normal project where the first attempt died for an unrelated reason,
|
|
206
198
|
* and an easy opt-out there would silence real research.
|
|
207
199
|
*
|
|
208
|
-
*
|
|
209
|
-
*
|
|
210
|
-
*
|
|
211
|
-
* an earlier wording did NOT take the `(none)` exit but drifted into writing code
|
|
212
|
-
* instead of listing paths — the deliverable-not-inputs failure the base prompt
|
|
213
|
-
* already forbids below this preamble. Blast radius is bounded: the gate fires
|
|
214
|
-
* only on a run that would otherwise have FAILED outright, so a mediocre recovered
|
|
215
|
-
* section is strictly better than the dead task it replaces — but if the drift
|
|
216
|
-
* reproduces, this preamble must restate the section's output contract, not just
|
|
217
|
-
* demand an answer.
|
|
200
|
+
* The blast radius is bounded: this gate fires only on a run that would otherwise
|
|
201
|
+
* have FAILED outright, so even a mediocre recovered section beats the dead task
|
|
202
|
+
* it replaces.
|
|
218
203
|
*/
|
|
219
204
|
const EMPTY_SECTION_PREAMBLE = 'STOP. Your previous attempt returned an EMPTY answer — zero characters. An empty '
|
|
220
205
|
+ 'response cannot be accepted, because it is indistinguishable from a worker that '
|
|
@@ -250,21 +235,16 @@ export async function runResearchWorker(spec, run, prior = []) {
|
|
|
250
235
|
contextWindow: run.contextWindow,
|
|
251
236
|
signal: run.signal,
|
|
252
237
|
spawn: run.spawn,
|
|
253
|
-
// ONE CELL PER WORKER
|
|
254
|
-
// the
|
|
255
|
-
//
|
|
256
|
-
//
|
|
257
|
-
//
|
|
258
|
-
//
|
|
259
|
-
// paid for the other. THE FOUR CELLS DO NOT SHIP IDENTICAL:
|
|
260
|
-
// `research:files` is `off` on a measured tie while the
|
|
261
|
-
// other three are `medium`, so this line changes what the
|
|
262
|
-
// FILES worker runs at for every default-mode user. The
|
|
263
|
-
// evidence is on each cell in reasoning.ts.
|
|
238
|
+
// ONE REASONING CELL PER WORKER, keyed on the spec's LABEL —
|
|
239
|
+
// the four workers ask four different questions and the four
|
|
240
|
+
// cells do not ship identical. `REASONING_DEFAULTS` in
|
|
241
|
+
// config/reasoning.ts is where each one's level lives, so this
|
|
242
|
+
// line decides what THIS worker runs at for a default-mode
|
|
243
|
+
// user.
|
|
264
244
|
thinking: run.thinkingFor(spec.label),
|
|
265
245
|
...(spec.tools ? { tools: spec.tools } : {}),
|
|
266
246
|
...(spec.extensions ? { extensions: spec.extensions } : {}),
|
|
267
|
-
// The three
|
|
247
|
+
// The three lever spreads that would otherwise sit here are the
|
|
268
248
|
// `research` row of WORKER_PROFILES (workers/worker-profiles.ts).
|
|
269
249
|
// Two facts still come from here, and only these two: which of
|
|
270
250
|
// the four workers is docs-capable (only it can be scaled), and
|
|
@@ -276,10 +256,8 @@ export async function runResearchWorker(spec, run, prior = []) {
|
|
|
276
256
|
env: run.leverEnv
|
|
277
257
|
},
|
|
278
258
|
// One line per DISCARDED attempt. The `done` line below reports
|
|
279
|
-
// the
|
|
280
|
-
//
|
|
281
|
-
// 8 minutes of burned compute recoverable only by subtracting
|
|
282
|
-
// its own wait+work from the start/done timestamps.
|
|
259
|
+
// the FINAL attempt only, so without these a worker that timed
|
|
260
|
+
// out twice and then answered logs exactly like a clean one.
|
|
283
261
|
onCarryForward: ci => {
|
|
284
262
|
run.logDebug?.(`${spec.label}: CARRY-FORWARD injected into attempt ${ci.attempt}`
|
|
285
263
|
+ ` (${ci.chars} chars onto a ${ci.promptCharsBefore}-char prompt)`);
|
|
@@ -301,8 +279,8 @@ export async function runResearchWorker(spec, run, prior = []) {
|
|
|
301
279
|
}
|
|
302
280
|
}));
|
|
303
281
|
let r = await runOnce();
|
|
304
|
-
// EMPTY-SECTION GATE
|
|
305
|
-
//
|
|
282
|
+
// EMPTY-SECTION GATE. A worker that returns zero bytes on a clean run
|
|
283
|
+
// would fail the whole task ("Research APIS worker produced no output"), which is
|
|
306
284
|
// exactly what an extremely simple task provokes: with nothing on disk to survey and
|
|
307
285
|
// no external symbol in play, silence is the correct answer and the run died on it.
|
|
308
286
|
// Retry ONCE — silence is genuinely ambiguous, and a worker that crashed before
|
|
@@ -396,11 +374,9 @@ export async function runResearchWorker(spec, run, prior = []) {
|
|
|
396
374
|
+ (r.restarts.length > 0 ?
|
|
397
375
|
` restarts=[${r.restarts.map(x => x.reason).join(',')}]`
|
|
398
376
|
: '')
|
|
399
|
-
//
|
|
400
|
-
//
|
|
401
|
-
//
|
|
402
|
-
// this the two are indistinguishable in the logs, and "0
|
|
403
|
-
// timeouts" cannot be traced to the half that earned it.
|
|
377
|
+
// A run with zero restarts was never killed; a run that restarted
|
|
378
|
+
// and salvaged WAS killed but kept its work. Without this flag the
|
|
379
|
+
// two are indistinguishable in the log.
|
|
404
380
|
+ (r.salvagedFromDiscardedAttempt ? ' salvaged=1' : '')
|
|
405
381
|
+ (r.stderr ? ` stderr=${r.stderr.slice(0, 300)}` : '')
|
|
406
382
|
+ (r.leakedToolCall ? ` leaked=${r.leakedToolCall.trim().slice(0, 80)}` : ''));
|
|
@@ -1,25 +1,24 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Resume gating & the honest resume banner.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* the run — the only defect was that nobody could tell. Two things follow.
|
|
4
|
+
* A host that is powered off mid-run produces dead air that looks exactly like a
|
|
5
|
+
* stall. Nothing is wrong with the run; the defect is that nobody can tell. Two
|
|
6
|
+
* things follow.
|
|
8
7
|
*
|
|
9
|
-
* (1) A restart-time resume can be automated
|
|
10
|
-
*
|
|
11
|
-
* turns that 10h of nothing into minutes.
|
|
8
|
+
* (1) A restart-time resume can be automated — a boot hook that runs
|
|
9
|
+
* `/task-auto-resume --unattended` (the flag auto-orchestrator.ts parses).
|
|
12
10
|
*
|
|
13
11
|
* (2) An automated resume must not resume everything. `RESUMABLE_STATES`
|
|
14
|
-
* deliberately includes `failed` and `cancelled
|
|
15
|
-
* /task-auto-resume has decided to continue; a boot hook has decided
|
|
16
|
-
* failed run stopped for a reason a power cycle does not clear, and
|
|
17
|
-
* it unattended burns the whole loop against the same wall.
|
|
18
|
-
* therefore covers in-flight states only (see
|
|
19
|
-
* the rest
|
|
12
|
+
* (task-types.ts) deliberately includes `failed` and `cancelled`, because a HUMAN
|
|
13
|
+
* typing /task-auto-resume has decided to continue; a boot hook has decided
|
|
14
|
+
* nothing. A failed run stopped for a reason a power cycle does not clear, and
|
|
15
|
+
* re-entering it unattended burns the whole loop against the same wall.
|
|
16
|
+
* Unattended resume therefore covers in-flight states only (see
|
|
17
|
+
* UNATTENDED_STATES), and refuses the rest BY NAME rather than silently doing
|
|
18
|
+
* nothing.
|
|
20
19
|
*
|
|
21
|
-
* The banner
|
|
22
|
-
*
|
|
20
|
+
* The banner says exactly what was observed and exactly what it does not tell
|
|
21
|
+
* you. All this process knows is
|
|
23
22
|
* when the AUTO file was last written — it cannot distinguish a stopped host
|
|
24
23
|
* from a hung child from a slow task, so it reports the gap and attributes no
|
|
25
24
|
* cause. It is equally careful about the tree: nothing is rolled back between
|
|
@@ -87,10 +87,11 @@ export interface MergedRepair {
|
|
|
87
87
|
verifyCommand?: string;
|
|
88
88
|
}
|
|
89
89
|
/**
|
|
90
|
-
* Collapse candidates by file — MANDATORY dedup:
|
|
91
|
-
*
|
|
92
|
-
*
|
|
93
|
-
* accumulate in
|
|
90
|
+
* Collapse candidates by file — MANDATORY dedup: two debts naming the same file
|
|
91
|
+
* must yield exactly ONE repair task naming both blamed tasks. Keyed on the
|
|
92
|
+
* NORMALISED path, so `./x.ts` and `x.ts` are one entry. First record wins for
|
|
93
|
+
* defect and command (they describe the same fault); blamed tasks accumulate in
|
|
94
|
+
* first-seen order.
|
|
94
95
|
*/
|
|
95
96
|
export declare function mergeRepairCandidates(candidates: RepairCandidate[]): MergedRepair[];
|
|
96
97
|
/** Machine-recognisable prefix, so a repair entry can be found in a plan again. */
|
|
@@ -112,11 +113,10 @@ export declare function parseRepairTitleFile(title: string): string | null;
|
|
|
112
113
|
*/
|
|
113
114
|
export declare function planHasRepairFor(titles: string[], file: string): boolean;
|
|
114
115
|
/**
|
|
115
|
-
* The extra scope fence a repair entry carries into refine.
|
|
116
|
-
*
|
|
117
|
-
*
|
|
118
|
-
*
|
|
119
|
-
* the debt failed on.
|
|
116
|
+
* The extra scope fence a repair entry carries into refine. A repair title names
|
|
117
|
+
* one file and one narrow defect, which refine will otherwise re-expand into
|
|
118
|
+
* "overhaul the test infrastructure". The fence pins the single editable file and
|
|
119
|
+
* pins the VERIFY to the exact command the debt failed on.
|
|
120
120
|
*/
|
|
121
121
|
export declare function buildRepairScopeFence(file: string, verifyCommand?: string): string;
|
|
122
122
|
export {};
|