@mjasnikovs/pi-task 0.38.29 → 0.38.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config/config.d.ts +70 -70
- package/dist/config/config.js +26 -35
- package/dist/config/extension-list.d.ts +6 -5
- package/dist/config/extension-list.js +3 -2
- package/dist/config/reasoning-args.d.ts +9 -7
- package/dist/config/reasoning-args.js +12 -10
- package/dist/config/reasoning.d.ts +44 -105
- package/dist/config/reasoning.js +27 -704
- package/dist/config/register.d.ts +34 -48
- package/dist/config/register.js +41 -51
- package/dist/config/tool-list.d.ts +16 -16
- package/dist/config/tool-list.js +1 -1
- package/dist/remote/bridge.d.ts +19 -10
- package/dist/remote/bridge.js +3 -2
- package/dist/remote/broadcast.js +3 -1
- package/dist/remote/events.js +12 -11
- package/dist/remote/history.d.ts +1 -1
- package/dist/remote/protocol.d.ts +6 -3
- package/dist/remote/protocol.js +2 -1
- package/dist/remote/push.d.ts +16 -16
- package/dist/remote/push.js +27 -27
- package/dist/remote/register.d.ts +3 -3
- package/dist/remote/register.js +17 -19
- package/dist/remote/server.d.ts +9 -8
- package/dist/remote/server.js +15 -14
- package/dist/remote/session-state.d.ts +5 -4
- package/dist/remote/session-state.js +8 -5
- package/dist/remote/sw.d.ts +7 -6
- package/dist/remote/sw.js +7 -6
- package/dist/remote/tailscale.d.ts +4 -2
- package/dist/remote/tailscale.js +4 -2
- package/dist/remote/ui-highlight.js +6 -5
- package/dist/remote/ui-render.js +4 -4
- package/dist/remote/ui-script.js +24 -24
- package/dist/remote/ui-styles.d.ts +1 -1
- package/dist/remote/ui-styles.js +10 -13
- package/dist/remote/ui-tools.js +9 -6
- package/dist/shared/child-extensions.d.ts +29 -17
- package/dist/shared/child-extensions.js +29 -17
- package/dist/shared/child-output.d.ts +30 -24
- package/dist/shared/child-output.js +25 -17
- package/dist/shared/child-process.d.ts +47 -40
- package/dist/shared/child-process.js +50 -59
- package/dist/shared/command-watchdog.d.ts +22 -16
- package/dist/shared/command-watchdog.js +28 -21
- package/dist/shared/fs-text.d.ts +16 -10
- package/dist/shared/fs-text.js +16 -10
- package/dist/shared/git-runner.d.ts +25 -25
- package/dist/shared/git-runner.js +25 -25
- package/dist/shared/leaked-tool-call.d.ts +17 -11
- package/dist/shared/leaked-tool-call.js +23 -15
- package/dist/shared/model-endpoint.d.ts +29 -16
- package/dist/shared/model-endpoint.js +33 -21
- package/dist/shared/pi-invocation.d.ts +7 -4
- package/dist/shared/pi-invocation.js +12 -7
- package/dist/shared/pkg-version.d.ts +13 -5
- package/dist/shared/pkg-version.js +13 -5
- package/dist/shared/reasoning-capability.d.ts +35 -24
- package/dist/shared/reasoning-capability.js +35 -24
- package/dist/shared/stream-watchdog.d.ts +60 -44
- package/dist/shared/stream-watchdog.js +62 -45
- package/dist/task/accept-debt.d.ts +41 -43
- package/dist/task/accept-debt.js +73 -65
- package/dist/task/api-synthesis.d.ts +24 -21
- package/dist/task/api-synthesis.js +32 -26
- package/dist/task/apis-contract.d.ts +32 -64
- package/dist/task/apis-contract.js +32 -64
- package/dist/task/artifact-closure.d.ts +27 -13
- package/dist/task/artifact-closure.js +95 -67
- package/dist/task/auto-commit.d.ts +46 -35
- package/dist/task/auto-commit.js +51 -38
- package/dist/task/auto-io.d.ts +45 -25
- package/dist/task/auto-io.js +57 -29
- package/dist/task/auto-orchestrator.d.ts +26 -24
- package/dist/task/auto-orchestrator.js +178 -162
- package/dist/task/auto-prompts.d.ts +36 -24
- package/dist/task/auto-prompts.js +40 -26
- package/dist/task/autofix-ledger.d.ts +27 -25
- package/dist/task/autofix-ledger.js +29 -26
- package/dist/task/batch-test-task.d.ts +20 -12
- package/dist/task/batch-test-task.js +67 -60
- package/dist/task/boot-probe.d.ts +60 -44
- package/dist/task/boot-probe.js +91 -72
- package/dist/task/cancel-input.d.ts +30 -16
- package/dist/task/cancel-input.js +20 -11
- package/dist/task/cancel-points.d.ts +27 -20
- package/dist/task/cancel-points.js +30 -22
- package/dist/task/child-runner.d.ts +46 -51
- package/dist/task/child-runner.js +48 -49
- package/dist/task/child-status.d.ts +23 -16
- package/dist/task/child-status.js +23 -16
- package/dist/task/clamp-output.js +12 -5
- package/dist/task/command-run.d.ts +31 -28
- package/dist/task/command-run.js +44 -35
- package/dist/task/command-shrink.d.ts +25 -18
- package/dist/task/command-shrink.js +37 -31
- package/dist/task/command-watchdog.d.ts +9 -6
- package/dist/task/command-watchdog.js +21 -15
- package/dist/task/context-attribution.d.ts +34 -26
- package/dist/task/context-attribution.js +34 -26
- package/dist/task/context-silence.d.ts +39 -29
- package/dist/task/context-silence.js +35 -25
- package/dist/task/context-usage.d.ts +16 -9
- package/dist/task/context-usage.js +16 -9
- package/dist/task/contracts.d.ts +8 -4
- package/dist/task/contracts.js +25 -17
- package/dist/task/coverage-loop.d.ts +22 -18
- package/dist/task/coverage-loop.js +35 -30
- package/dist/task/critique-probes.d.ts +13 -14
- package/dist/task/critique-probes.js +50 -39
- package/dist/task/debug-log.d.ts +13 -5
- package/dist/task/debug-log.js +32 -20
- package/dist/task/decompose-fidelity.d.ts +11 -9
- package/dist/task/decompose-fidelity.js +38 -33
- package/dist/task/decompose-granularity.d.ts +41 -38
- package/dist/task/decompose-granularity.js +41 -38
- package/dist/task/deep-render-check.d.ts +22 -14
- package/dist/task/deep-render-check.js +40 -31
- package/dist/task/dropped-input.d.ts +12 -7
- package/dist/task/dropped-input.js +5 -2
- package/dist/task/enforce-attribution.d.ts +38 -47
- package/dist/task/enforce-attribution.js +46 -52
- package/dist/task/enforce-guidelines.d.ts +31 -20
- package/dist/task/enforce-guidelines.js +32 -21
- package/dist/task/enrichment.d.ts +7 -2
- package/dist/task/enrichment.js +26 -14
- package/dist/task/env-notes.d.ts +16 -7
- package/dist/task/env-notes.js +48 -31
- package/dist/task/env-template-closure.d.ts +4 -4
- package/dist/task/env-template-closure.js +42 -34
- package/dist/task/external-context.d.ts +28 -21
- package/dist/task/external-context.js +17 -12
- package/dist/task/failure-classifier.d.ts +4 -5
- package/dist/task/failure-classifier.js +6 -7
- package/dist/task/file-inventory.d.ts +15 -11
- package/dist/task/file-inventory.js +25 -22
- package/dist/task/final-gate-fix.d.ts +74 -86
- package/dist/task/final-gate-fix.js +97 -116
- package/dist/task/final-gate-progress.d.ts +29 -46
- package/dist/task/final-gate-progress.js +40 -51
- package/dist/task/final-gate.d.ts +64 -97
- package/dist/task/final-gate.js +192 -199
- package/dist/task/fix-child.d.ts +21 -27
- package/dist/task/fix-child.js +21 -27
- package/dist/task/foreign-path.d.ts +6 -5
- package/dist/task/foreign-path.js +0 -0
- package/dist/task/frozen-conflict.d.ts +9 -10
- package/dist/task/frozen-conflict.js +61 -64
- package/dist/task/frozen-path-guard.d.ts +35 -14
- package/dist/task/frozen-path-guard.js +56 -39
- package/dist/task/gate-child.d.ts +27 -28
- package/dist/task/gate-child.js +36 -35
- package/dist/task/gate-deps.d.ts +34 -27
- package/dist/task/gate-deps.js +169 -159
- package/dist/task/gate-tally.d.ts +77 -80
- package/dist/task/gate-tally.js +65 -68
- package/dist/task/git-state-guard.d.ts +15 -11
- package/dist/task/git-state-guard.js +76 -66
- package/dist/task/impl-widget.d.ts +25 -16
- package/dist/task/impl-widget.js +27 -17
- package/dist/task/implementation-thinking.d.ts +33 -31
- package/dist/task/implementation-thinking.js +5 -6
- package/dist/task/implementation-turn.d.ts +34 -31
- package/dist/task/implementation-turn.js +29 -27
- package/dist/task/inline-markdown.d.ts +20 -7
- package/dist/task/inline-markdown.js +15 -6
- package/dist/task/launch-config-gap.js +25 -39
- package/dist/task/launch-contract.d.ts +18 -21
- package/dist/task/launch-contract.js +28 -30
- package/dist/task/launch-manifest.d.ts +6 -2
- package/dist/task/launch-manifest.js +35 -34
- package/dist/task/ledger.js +16 -14
- package/dist/task/lint-fix.d.ts +6 -8
- package/dist/task/lint-fix.js +67 -69
- package/dist/task/loop-detector.d.ts +9 -8
- package/dist/task/loop-detector.js +16 -12
- package/dist/task/mid-run-input.d.ts +17 -15
- package/dist/task/mid-run-input.js +17 -15
- package/dist/task/orchestrator.d.ts +24 -28
- package/dist/task/orchestrator.js +62 -64
- package/dist/task/orientation.d.ts +18 -23
- package/dist/task/orientation.js +24 -31
- package/dist/task/owned-freeze-conflict.d.ts +21 -20
- package/dist/task/owned-freeze-conflict.js +52 -85
- package/dist/task/owned-freeze-reassign.d.ts +40 -60
- package/dist/task/owned-freeze-reassign.js +41 -61
- package/dist/task/parsers.d.ts +4 -2
- package/dist/task/parsers.js +4 -4
- package/dist/task/phases.d.ts +41 -48
- package/dist/task/phases.js +179 -248
- package/dist/task/plan-io.d.ts +6 -7
- package/dist/task/plan-io.js +6 -7
- package/dist/task/plan-orchestrator.d.ts +10 -8
- package/dist/task/plan-orchestrator.js +14 -10
- package/dist/task/plan-prompts.d.ts +6 -5
- package/dist/task/plan-prompts.js +6 -5
- package/dist/task/plan-readonly.d.ts +4 -5
- package/dist/task/plan-readonly.js +4 -5
- package/dist/task/plan-rounds.d.ts +17 -29
- package/dist/task/plan-rounds.js +21 -34
- package/dist/task/plan-session.d.ts +58 -72
- package/dist/task/plan-session.js +61 -83
- package/dist/task/probe-gaming.d.ts +28 -27
- package/dist/task/probe-gaming.js +0 -0
- package/dist/task/prohibition-probe.d.ts +14 -16
- package/dist/task/prompts.d.ts +3 -4
- package/dist/task/prompts.js +17 -26
- package/dist/task/qa-transcript.d.ts +15 -22
- package/dist/task/qa-transcript.js +15 -21
- package/dist/task/question-box.d.ts +17 -13
- package/dist/task/question-box.js +19 -15
- package/dist/task/question-dedup.d.ts +6 -7
- package/dist/task/question-dedup.js +13 -14
- package/dist/task/question-dialog.d.ts +22 -32
- package/dist/task/question-dialog.js +22 -32
- package/dist/task/question-source.d.ts +18 -44
- package/dist/task/question-source.js +22 -51
- package/dist/task/refuted-constraint.d.ts +11 -31
- package/dist/task/refuted-constraint.js +27 -51
- package/dist/task/regenerable-artifacts.d.ts +12 -31
- package/dist/task/regenerable-artifacts.js +12 -31
- package/dist/task/render-check.d.ts +11 -22
- package/dist/task/render-check.js +33 -46
- package/dist/task/repo-health-check.d.ts +10 -14
- package/dist/task/repo-health-check.js +17 -23
- package/dist/task/requirements.d.ts +38 -71
- package/dist/task/requirements.js +78 -126
- package/dist/task/research-fanout-budget.d.ts +51 -88
- package/dist/task/research-fanout-budget.js +51 -88
- package/dist/task/research-worker.d.ts +29 -39
- package/dist/task/research-worker.js +37 -61
- package/dist/task/resume-gap.d.ts +14 -15
- package/dist/task/root-cause-repair.d.ts +9 -9
- package/dist/task/root-cause-repair.js +28 -40
- package/dist/task/run-bracket.d.ts +10 -13
- package/dist/task/run-end.d.ts +12 -22
- package/dist/task/run-end.js +8 -16
- package/dist/task/run-final-gate.d.ts +19 -21
- package/dist/task/run-final-gate.js +62 -80
- package/dist/task/runner-globs.d.ts +12 -13
- package/dist/task/runner-globs.js +12 -13
- package/dist/task/runner-resolve.d.ts +9 -9
- package/dist/task/runner-resolve.js +22 -23
- package/dist/task/script-escape.d.ts +10 -12
- package/dist/task/script-escape.js +13 -14
- package/dist/task/serve-entry.d.ts +1 -1
- package/dist/task/serve-entry.js +22 -25
- package/dist/task/service-blocks.js +4 -2
- package/dist/task/shipped-source.d.ts +11 -29
- package/dist/task/shipped-source.js +11 -29
- package/dist/task/skip-escape.js +10 -14
- package/dist/task/spec-urls.d.ts +26 -65
- package/dist/task/spec-urls.js +26 -65
- package/dist/task/spec-validation.d.ts +17 -20
- package/dist/task/spec-validation.js +17 -20
- package/dist/task/stall-detector.d.ts +23 -30
- package/dist/task/stall-detector.js +23 -30
- package/dist/task/stream-watchdog.d.ts +14 -12
- package/dist/task/stream-watchdog.js +14 -12
- package/dist/task/substitution-probe.d.ts +17 -20
- package/dist/task/substitution-probe.js +17 -20
- package/dist/task/task-gates.d.ts +36 -41
- package/dist/task/task-gates.js +95 -106
- package/dist/task/task-io.d.ts +4 -4
- package/dist/task/task-io.js +4 -4
- package/dist/task/task-parsers.js +4 -3
- package/dist/task/task-provenance.d.ts +2 -2
- package/dist/task/task-provenance.js +11 -13
- package/dist/task/task-types.d.ts +4 -3
- package/dist/task/terminal-outcome.d.ts +14 -16
- package/dist/task/terminal-outcome.js +12 -14
- package/dist/task/test-assembly.d.ts +13 -20
- package/dist/task/test-assembly.js +13 -20
- package/dist/task/timings.d.ts +5 -3
- package/dist/task/timings.js +5 -3
- package/dist/task/title-label.d.ts +9 -4
- package/dist/task/title-label.js +9 -4
- package/dist/task/type-only-answer.d.ts +44 -52
- package/dist/task/type-only-answer.js +44 -52
- package/dist/task/unfailable-command.d.ts +18 -24
- package/dist/task/unfailable-command.js +21 -27
- package/dist/task/unknown-routing.d.ts +10 -4
- package/dist/task/unknown-routing.js +10 -4
- package/dist/task/user-directives.d.ts +5 -8
- package/dist/task/user-directives.js +5 -8
- package/dist/task/verify-quality.d.ts +18 -22
- package/dist/task/verify-quality.js +45 -46
- package/dist/task/verify-reconcile.d.ts +15 -10
- package/dist/task/verify-reconcile.js +45 -43
- package/dist/task/verify-resolution.d.ts +24 -20
- package/dist/task/verify-resolution.js +51 -50
- package/dist/task/verify-work.d.ts +59 -66
- package/dist/task/verify-work.js +101 -138
- package/dist/task/widget.d.ts +15 -14
- package/dist/task/widget.js +22 -17
- package/dist/task/wiring-claims.d.ts +25 -32
- package/dist/task/wiring-claims.js +30 -35
- package/dist/task/write-guard.d.ts +39 -39
- package/dist/task/write-guard.js +48 -51
- package/dist/task/yolo.d.ts +34 -30
- package/dist/task/yolo.js +42 -37
- package/dist/workers/abstention.d.ts +21 -41
- package/dist/workers/abstention.js +27 -48
- package/dist/workers/brave-search.d.ts +4 -3
- package/dist/workers/brave-search.js +5 -2
- package/dist/workers/brave-warning.d.ts +7 -4
- package/dist/workers/brave-warning.js +19 -7
- package/dist/workers/ddg-search.d.ts +6 -6
- package/dist/workers/ddg-search.js +18 -12
- package/dist/workers/docs-cache.js +5 -2
- package/dist/workers/docs-chunk.d.ts +30 -37
- package/dist/workers/docs-chunk.js +37 -41
- package/dist/workers/docs-core.d.ts +28 -44
- package/dist/workers/docs-core.js +25 -44
- package/dist/workers/docs-index.js +4 -3
- package/dist/workers/docs-lookup.d.ts +15 -22
- package/dist/workers/docs-lookup.js +12 -21
- package/dist/workers/docs-project.d.ts +15 -9
- package/dist/workers/docs-project.js +17 -10
- package/dist/workers/docs-resolve.d.ts +19 -20
- package/dist/workers/docs-resolve.js +35 -32
- package/dist/workers/docs-retrieve.d.ts +5 -6
- package/dist/workers/docs-retrieve.js +18 -15
- package/dist/workers/exa-search.d.ts +9 -6
- package/dist/workers/exa-search.js +23 -12
- package/dist/workers/fetch-core.d.ts +13 -16
- package/dist/workers/fetch-core.js +23 -23
- package/dist/workers/focused-extractor.d.ts +12 -12
- package/dist/workers/focused-extractor.js +16 -19
- package/dist/workers/html-clean.js +24 -14
- package/dist/workers/http-request.d.ts +28 -20
- package/dist/workers/http-request.js +22 -17
- package/dist/workers/npm-version.d.ts +28 -11
- package/dist/workers/npm-version.js +24 -15
- package/dist/workers/phantom-imports.d.ts +15 -12
- package/dist/workers/phantom-imports.js +30 -24
- package/dist/workers/pi-worker-core.d.ts +69 -71
- package/dist/workers/pi-worker-core.js +100 -109
- package/dist/workers/pi-worker-docs.d.ts +24 -19
- package/dist/workers/pi-worker-docs.js +67 -76
- package/dist/workers/pi-worker-fetch.d.ts +7 -3
- package/dist/workers/pi-worker-fetch.js +27 -19
- package/dist/workers/pi-worker-search.js +12 -8
- package/dist/workers/pi-worker.d.ts +9 -4
- package/dist/workers/pi-worker.js +21 -14
- package/dist/workers/reasoning-warning.d.ts +18 -17
- package/dist/workers/reasoning-warning.js +22 -20
- package/dist/workers/research-cache.js +50 -78
- package/dist/workers/search-core.js +7 -5
- package/dist/workers/search-types.d.ts +10 -9
- package/dist/workers/search-types.js +9 -8
- package/dist/workers/session-hint.d.ts +13 -14
- package/dist/workers/session-hint.js +8 -9
- package/dist/workers/shared.d.ts +21 -25
- package/dist/workers/shared.js +0 -0
- package/dist/workers/single-read-extension.d.ts +14 -7
- package/dist/workers/single-read-extension.js +14 -7
- package/dist/workers/single-read-guard.d.ts +25 -28
- package/dist/workers/single-read-guard.js +32 -32
- package/dist/workers/typeonly-log.d.ts +12 -9
- package/dist/workers/typeonly-log.js +29 -33
- package/dist/workers/worker-channels.d.ts +15 -23
- package/dist/workers/worker-channels.js +15 -23
- package/dist/workers/worker-failure.d.ts +38 -46
- package/dist/workers/worker-failure.js +31 -39
- package/dist/workers/worker-kill.d.ts +25 -26
- package/dist/workers/worker-kill.js +16 -19
- package/dist/workers/worker-profiles.d.ts +43 -53
- package/dist/workers/worker-profiles.js +30 -38
- package/package.json +10 -8
|
@@ -6,44 +6,48 @@ import { type WorkerGuardOverride, type WorkerGuardPolicy, type WorkerPolicyInpu
|
|
|
6
6
|
* command could be cited from. `pi-worker-docs` (the primary), `read` and `grep`
|
|
7
7
|
* (project source), and the web escalations `pi-worker-search`/`pi-worker-fetch`.
|
|
8
8
|
*
|
|
9
|
-
* `ls` and `find` are deliberately EXCLUDED: they return file
|
|
10
|
-
* and APIS owns symbols by name only, never paths (
|
|
11
|
-
* enumeration cannot verify a
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
9
|
+
* `ls` and `find` are deliberately EXCLUDED: they return file and directory
|
|
10
|
+
* NAMES, and APIS owns symbols by name only, never paths (see
|
|
11
|
+
* RESEARCH_APIS_PROMPT in prompts.ts). Bare enumeration cannot verify a
|
|
12
|
+
* signature, so a worker that fabricates its section from memory cannot launder
|
|
13
|
+
* itself grounded by calling `ls` once — "one trivial `ls` then fabricate the
|
|
14
|
+
* rest" still leaves groundingRetrievalCount at 0.
|
|
15
|
+
*
|
|
16
|
+
* The set is DERIVED from WORKER_CHANNELS (worker-channels.ts), not hand-kept.
|
|
17
|
+
* Re-exported here only so worker-channels.test.ts can assert this module hands
|
|
18
|
+
* back the same predicate.
|
|
15
19
|
*/
|
|
16
20
|
export { isGroundingRetrieval } from './worker-channels.js';
|
|
17
21
|
/**
|
|
18
22
|
* Does this partial output carry ANSWER CONTENT, or is it the model clearing its
|
|
19
23
|
* throat?
|
|
20
24
|
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
* attempts and salvage shipped this as the section:
|
|
25
|
+
* Keeping the LONGEST partial is not the same question, and it has an obvious
|
|
26
|
+
* failure: a preamble sentence like
|
|
24
27
|
*
|
|
25
28
|
* "Now let me get more details on the specific APIs and components I need:"
|
|
26
29
|
*
|
|
27
|
-
*
|
|
28
|
-
* nothing. Both trials scored 2 entries and DEGRADED, against 22 and 5 for the
|
|
29
|
-
* same fixtures in baseline.
|
|
30
|
+
* beats an empty string on length and carries nothing.
|
|
30
31
|
*
|
|
31
32
|
* A research worker's answer is a list of lines that each name something and
|
|
32
|
-
* describe it. The test is therefore structural, not lexical: at least
|
|
33
|
-
* that look like entries — a name, then a gap, then a description. Prose wraps
|
|
34
|
-
*
|
|
33
|
+
* describe it. The test is therefore structural, not lexical: at least TWO lines
|
|
34
|
+
* that look like entries — a name, then a gap, then a description. Prose wraps at
|
|
35
|
+
* no particular column and does not repeat that shape; the sentence above scores
|
|
36
|
+
* zero entry lines.
|
|
35
37
|
*/
|
|
36
38
|
export declare function hasAnswerContent(text: string): boolean;
|
|
37
39
|
/**
|
|
38
40
|
* Is ONE line an entry — a name, a gap, then a description — rather than prose?
|
|
39
41
|
*
|
|
40
|
-
* Split out of `hasAnswerContent` so the same rule can decide what a line IS,
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
42
|
+
* Split out of `hasAnswerContent` so the same rule can decide what a line IS, not
|
|
43
|
+
* just how many of them there are — a reader of a FILES section needs the same
|
|
44
|
+
* test, and its own idea of an entry would count a preamble sentence or a leaked
|
|
45
|
+
* `</tool_call>` as one.
|
|
44
46
|
*
|
|
45
|
-
*
|
|
46
|
-
* spaced dash
|
|
47
|
+
* A leading `-`, `*`, `•` or `1.`/`1)` bullet is stripped first. What remains must
|
|
48
|
+
* hold a two-space gap or a spaced dash and must NOT end in `.` or `:`. Prose
|
|
49
|
+
* wraps at no particular column, so it carries neither; when it does carry one, it
|
|
50
|
+
* ends in punctuation and an entry does not.
|
|
47
51
|
*/
|
|
48
52
|
export declare function isEntryLine(raw: string): boolean;
|
|
49
53
|
/**
|
|
@@ -67,7 +71,7 @@ export interface RunWorkerInput {
|
|
|
67
71
|
/** Called for each tool execution start and text-writing event inside the worker. */
|
|
68
72
|
onLine?: (line: string) => void;
|
|
69
73
|
/** Called when a tool call FINISHES, with its (truncatable) result — lets a caller
|
|
70
|
-
* log tool OUTPUTS, not just the command
|
|
74
|
+
* log tool OUTPUTS, not just the command. */
|
|
71
75
|
onToolResult?: (result: {
|
|
72
76
|
name: string;
|
|
73
77
|
isError: boolean;
|
|
@@ -85,34 +89,31 @@ export interface RunWorkerInput {
|
|
|
85
89
|
* The worker child's context window in tokens, or `'unknown'` when the
|
|
86
90
|
* caller genuinely has none.
|
|
87
91
|
*
|
|
88
|
-
* REQUIRED, and required for the same reason `profile` below is.
|
|
89
|
-
*
|
|
90
|
-
*
|
|
91
|
-
*
|
|
92
|
-
*
|
|
93
|
-
*
|
|
94
|
-
*
|
|
95
|
-
*
|
|
96
|
-
* exists and did not trip.
|
|
92
|
+
* REQUIRED, and required for the same reason `profile` below is. pi's event
|
|
93
|
+
* stream carries NO window — the string `context_usage` appears nowhere in any
|
|
94
|
+
* installed @earendil-works package, and its only usage-bearing JSON event is
|
|
95
|
+
* `message_update` — so this parameter is the only source there is. Left
|
|
96
|
+
* optional, a caller that omits it leaves `noteContext` seeing 0, and
|
|
97
|
+
* `StallDetector`'s CONTEXT CHURN rule is gated on a positive window, so the
|
|
98
|
+
* rule silently does not exist. That reads exactly like a rule that exists and
|
|
99
|
+
* did not trip.
|
|
97
100
|
*
|
|
98
|
-
* WHY A WORD AND NOT `0` OR `null`. Both of those are what a caller types
|
|
99
|
-
*
|
|
100
|
-
*
|
|
101
|
-
*
|
|
101
|
+
* WHY A WORD AND NOT `0` OR `null`. Both of those are what a caller types when
|
|
102
|
+
* it has not thought about the question, and both disarm the rule silently.
|
|
103
|
+
* `'unknown'` cannot be typed by accident, is greppable, and shows up in a diff
|
|
104
|
+
* as a decision.
|
|
102
105
|
*
|
|
103
106
|
* Two consumers read it: the churn rule, and the caller's progress bar, which
|
|
104
|
-
* shows a bare token count without a window. Both degrade
|
|
105
|
-
* on `'unknown'`.
|
|
107
|
+
* shows a bare token count without a window. Both degrade on `'unknown'`.
|
|
106
108
|
*/
|
|
107
109
|
contextWindow: number | 'unknown';
|
|
108
110
|
/**
|
|
109
111
|
* WHICH KIND of worker child this is — the whole guard policy, in one word.
|
|
110
112
|
*
|
|
111
|
-
* REQUIRED, and required on purpose.
|
|
112
|
-
*
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
* children without anyone deciding it should. See worker-profiles.ts.
|
|
113
|
+
* REQUIRED, and required on purpose. As independent optionals, a caller that
|
|
114
|
+
* named none of the guard knobs still got a full policy and nobody could see
|
|
115
|
+
* which one — so a child can end up running the strictest wall clock of the
|
|
116
|
+
* three without anyone deciding it should. See worker-profiles.ts.
|
|
116
117
|
*/
|
|
117
118
|
profile: WorkerProfileId;
|
|
118
119
|
/**
|
|
@@ -121,10 +122,11 @@ export interface RunWorkerInput {
|
|
|
121
122
|
*/
|
|
122
123
|
policyInputs?: WorkerPolicyInputs;
|
|
123
124
|
/**
|
|
124
|
-
* Whole guard rows laid over the profile's. TESTS AND
|
|
125
|
-
*
|
|
126
|
-
*
|
|
127
|
-
*
|
|
125
|
+
* Whole guard rows laid over the profile's. TESTS AND HARNESSES ONLY — an
|
|
126
|
+
* override at a production call site is the hand-picked subset this design
|
|
127
|
+
* exists to stop. `worker-profiles.test.ts` enforces it: its "no production
|
|
128
|
+
* source file passes an `override` to runWorker" test scans src/ for a leading
|
|
129
|
+
* `override:` and fails on any hit.
|
|
128
130
|
*/
|
|
129
131
|
override?: WorkerGuardOverride;
|
|
130
132
|
/**
|
|
@@ -133,9 +135,9 @@ export interface RunWorkerInput {
|
|
|
133
135
|
*
|
|
134
136
|
* WHY: asserting that a profile RESOLVES correctly proves nothing about
|
|
135
137
|
* whether runWorker then READS it correctly — a rewiring that turns "0 means
|
|
136
|
-
* off" into "0 means on" leaves every profile assertion green. This hook
|
|
137
|
-
*
|
|
138
|
-
*
|
|
138
|
+
* off" into "0 means on" leaves every profile assertion green. This hook lets
|
|
139
|
+
* a test drive the REAL call site and read back the REAL policy instead of
|
|
140
|
+
* re-typing the table; worker-profiles.test.ts is where those assertions live.
|
|
139
141
|
*/
|
|
140
142
|
onPolicy?: (policy: WorkerGuardPolicy) => void;
|
|
141
143
|
/** Backoff sleep, injectable so tests don't wait out the real delays. */
|
|
@@ -168,12 +170,10 @@ export interface RunWorkerInput {
|
|
|
168
170
|
*
|
|
169
171
|
* WHY: every restart branch below throws away a whole attempt's wall clock
|
|
170
172
|
* along with its text, and `waitMs`/`workMs` describe the FINAL attempt only.
|
|
171
|
-
* With no hook here
|
|
172
|
-
*
|
|
173
|
-
*
|
|
174
|
-
*
|
|
175
|
-
* was only recoverable by subtracting reported wait+work from the timestamps
|
|
176
|
-
* of the `start` and `done` lines around it.
|
|
173
|
+
* With no hook here a discarded attempt is structurally invisible — the worker
|
|
174
|
+
* still returns `exitCode` 0 and reads as a clean success, and the lost time is
|
|
175
|
+
* recoverable only by subtracting the reported wait+work from the timestamps
|
|
176
|
+
* around the call.
|
|
177
177
|
*/
|
|
178
178
|
onRestart?: (restart: WorkerRestart) => void;
|
|
179
179
|
}
|
|
@@ -200,13 +200,9 @@ export interface WorkerRestart {
|
|
|
200
200
|
* thrown away.
|
|
201
201
|
*
|
|
202
202
|
* Recorded whatever `carryForward` says, because the DISCARD is the thing a
|
|
203
|
-
* reader cannot otherwise see.
|
|
204
|
-
*
|
|
205
|
-
*
|
|
206
|
-
* run threw away a finished answer" print identically. Measured on the
|
|
207
|
-
* ad-hoc `pi-worker` corpus: T024 lost an attempt to a dropped model socket
|
|
208
|
-
* at 275s and returned 52 chars over 620s; nothing in the run said whether
|
|
209
|
-
* those 275s held anything.
|
|
203
|
+
* reader cannot otherwise see. Without it, "the guards worked and the run
|
|
204
|
+
* returned almost nothing" and "the guards worked and the run threw away a
|
|
205
|
+
* finished answer" print identically.
|
|
210
206
|
*
|
|
211
207
|
* It is an OBSERVATION, not a decision: harvesting into `salvage` is still
|
|
212
208
|
* gated on the profile, and this number changes no behaviour.
|
|
@@ -222,9 +218,9 @@ export interface RunWorkerResult {
|
|
|
222
218
|
* The provider-reported cause when the model turn itself failed (disconnect,
|
|
223
219
|
* fetch failed, 5xx after pi's own retries): pi delivers it as an assistant
|
|
224
220
|
* message with stopReason "error" and EMPTY text, exit code 0. Phase children
|
|
225
|
-
*
|
|
226
|
-
*
|
|
227
|
-
*
|
|
221
|
+
* surface this through child-runner.ts; without it a swallowed provider error
|
|
222
|
+
* reaches the caller as an indistinguishable empty answer and gets reported as
|
|
223
|
+
* the useless "produced no output".
|
|
228
224
|
* Only meaningful when `text` is empty: a turn that produced text after pi
|
|
229
225
|
* recovered is a success, and the first-error capture must not relabel it.
|
|
230
226
|
*/
|
|
@@ -349,9 +345,9 @@ export interface RunWorkerResult {
|
|
|
349
345
|
* commandTimeoutHint, which tells the model in as many words to bound its
|
|
350
346
|
* command; a SECOND hang means it ignored an explicit instruction, and a third
|
|
351
347
|
* means it ignored it twice. Giving a non-complying child the full ceiling again
|
|
352
|
-
*
|
|
353
|
-
*
|
|
354
|
-
*
|
|
348
|
+
* makes the worst case three times the ceiling, resting entirely on the model
|
|
349
|
+
* obeying prose. Halving bounds it at under twice the ceiling while costing a
|
|
350
|
+
* complying child nothing.
|
|
355
351
|
*
|
|
356
352
|
* `priorHangs` counts watchdog kills specifically, NOT total restarts — the
|
|
357
353
|
* restart budget is shared with loop kills, and a child restarted for LOOPING
|
|
@@ -359,8 +355,9 @@ export interface RunWorkerResult {
|
|
|
359
355
|
* the full ceiling. Only a hang after a hang is defiance.
|
|
360
356
|
*
|
|
361
357
|
* Floored at 30s so repeated halving cannot shrink the ceiling to something no
|
|
362
|
-
* real command could finish inside — but
|
|
363
|
-
*
|
|
358
|
+
* real command could finish inside — but the floor is `min(base, 30s)`, never
|
|
359
|
+
* above the configured ceiling, so a caller asking for 10s keeps 10s at every
|
|
360
|
+
* hang count. A base of 0 or less disables the watchdog and stays 0.
|
|
364
361
|
*/
|
|
365
362
|
export declare function commandCeilingForAttempt(baseMs: number, priorHangs: number): number;
|
|
366
363
|
/**
|
|
@@ -377,7 +374,8 @@ interface RestartState {
|
|
|
377
374
|
timedOut: boolean;
|
|
378
375
|
modelError?: string;
|
|
379
376
|
leaked: string | null;
|
|
380
|
-
/** The cap this attempt actually died against
|
|
377
|
+
/** The cap this attempt actually died against, not the configured one:
|
|
378
|
+
* `extend`/`progress` can push the deadline out during the attempt. */
|
|
381
379
|
effectiveCapMs: number;
|
|
382
380
|
/** The child's tool string, which decides whether its edits can persist. */
|
|
383
381
|
tools: string;
|
|
@@ -10,15 +10,9 @@ import { detectLeakedToolCall, leakedToolCallHint, MAX_LEAK_RETRIES } from '../s
|
|
|
10
10
|
import { discoverModelEndpoints, probeModelEndpoints } from '../shared/model-endpoint.js';
|
|
11
11
|
import { streamStallHint } from '../shared/stream-watchdog.js';
|
|
12
12
|
import { classifyWorkerFailure } from './worker-failure.js';
|
|
13
|
-
import { CARRY_FORWARD_IDS } from './worker-kill.js';
|
|
13
|
+
import { CARRY_FORWARD_IDS, RESTART_ORDER } from './worker-kill.js';
|
|
14
14
|
import { applyOverride, WORKER_PROFILES } from './worker-profiles.js';
|
|
15
|
-
|
|
16
|
-
// buffering the assistant text and flushing on exit. That matters for the
|
|
17
|
-
// wait/work timing split: in text mode the first stdout chunk only arrives at
|
|
18
|
-
// the very end, so onFirstByte fires moments before close and workMs is
|
|
19
|
-
// effectively zero. With JSON events the first byte lands as soon as the
|
|
20
|
-
// model starts producing — making waitMs the real queue/cold-start cost and
|
|
21
|
-
// workMs the real generation+tool-call cost.
|
|
15
|
+
/** The tool whitelist a caller gets when it names none. */
|
|
22
16
|
const DEFAULT_TOOLS = 'read,grep,find,ls';
|
|
23
17
|
/**
|
|
24
18
|
* The one place `'unknown'` becomes the 0 both consumers already treat as
|
|
@@ -33,18 +27,19 @@ function contextWindowTokens(cw) {
|
|
|
33
27
|
* command could be cited from. `pi-worker-docs` (the primary), `read` and `grep`
|
|
34
28
|
* (project source), and the web escalations `pi-worker-search`/`pi-worker-fetch`.
|
|
35
29
|
*
|
|
36
|
-
* `ls` and `find` are deliberately EXCLUDED: they return file
|
|
37
|
-
* and APIS owns symbols by name only, never paths (
|
|
38
|
-
* enumeration cannot verify a
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
30
|
+
* `ls` and `find` are deliberately EXCLUDED: they return file and directory
|
|
31
|
+
* NAMES, and APIS owns symbols by name only, never paths (see
|
|
32
|
+
* RESEARCH_APIS_PROMPT in prompts.ts). Bare enumeration cannot verify a
|
|
33
|
+
* signature, so a worker that fabricates its section from memory cannot launder
|
|
34
|
+
* itself grounded by calling `ls` once — "one trivial `ls` then fabricate the
|
|
35
|
+
* rest" still leaves groundingRetrievalCount at 0.
|
|
36
|
+
*
|
|
37
|
+
* The set is DERIVED from WORKER_CHANNELS (worker-channels.ts), not hand-kept.
|
|
38
|
+
* Re-exported here only so worker-channels.test.ts can assert this module hands
|
|
39
|
+
* back the same predicate.
|
|
42
40
|
*/
|
|
43
|
-
// The grounding set is derived from WORKER_CHANNELS (worker-channels.ts), not
|
|
44
|
-
// hand-kept — this was a second copy of the four tool names. Re-exported because
|
|
45
|
-
// several call sites and tests import it from here.
|
|
46
41
|
export { isGroundingRetrieval } from './worker-channels.js';
|
|
47
|
-
// RESEARCH_WORKER_TIMEOUT_MS and STALL_AFTER_MS live on the profile table
|
|
42
|
+
// RESEARCH_WORKER_TIMEOUT_MS and STALL_AFTER_MS live on the profile table
|
|
48
43
|
// (worker-profiles.ts): they are the default VALUES of two guard rows, and a
|
|
49
44
|
// default that lives apart from the table stating it is a second place to look.
|
|
50
45
|
/**
|
|
@@ -59,49 +54,44 @@ const WORKER_TIMEOUT_HINT = '[SYSTEM NOTE: Your previous attempt ran out of time
|
|
|
59
54
|
/**
|
|
60
55
|
* How much of a discarded attempt's answer is carried into the next one.
|
|
61
56
|
*
|
|
62
|
-
* A restart
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
* lookups, 5 of 5 workers burned the FULL restart budget, because every attempt
|
|
67
|
-
* re-read the same files against the same clock and died in the same place.
|
|
57
|
+
* A restart that hands the re-spawn nothing but a hint is why WORKER_TIMEOUT_HINT
|
|
58
|
+
* above can tell a worker "do not re-explore ground you have already covered"
|
|
59
|
+
* while giving it no record of what that ground was. It cannot comply, so it
|
|
60
|
+
* re-reads the same files against the same clock and dies in the same place.
|
|
68
61
|
*
|
|
69
|
-
* Carrying the partial
|
|
70
|
-
*
|
|
71
|
-
*
|
|
72
|
-
*
|
|
73
|
-
*
|
|
74
|
-
* framed as findings to VERIFY-or-DROP rather than as settled fact.
|
|
62
|
+
* Carrying the partial forward is what lets a restart converge instead of repeat.
|
|
63
|
+
* The risk is real: a half-written or speculative entry, replayed under "already
|
|
64
|
+
* established", is how a fabrication gets laundered into a final answer. That is
|
|
65
|
+
* why `formatCarryForward` frames it as findings to VERIFY-or-DROP rather than as
|
|
66
|
+
* settled fact.
|
|
75
67
|
*/
|
|
76
68
|
const CARRY_FORWARD_LIMIT = 24_000;
|
|
77
69
|
/**
|
|
78
70
|
* Restart reasons whose partial output is worth keeping.
|
|
79
71
|
*
|
|
80
|
-
*
|
|
81
|
-
*
|
|
82
|
-
*
|
|
83
|
-
*
|
|
84
|
-
*
|
|
72
|
+
* Exactly four, derived from WORKER_KILLS: `command-timeout`, `stream-stall`,
|
|
73
|
+
* `worker-timeout` and `connection-error` all discard work the model genuinely
|
|
74
|
+
* did. A loop kill and a leaked tool call do not — the first is by definition the
|
|
75
|
+
* same call repeated, the second is malformed protocol text, and replaying either
|
|
76
|
+
* would feed the failure back to itself.
|
|
85
77
|
*/
|
|
86
78
|
const CARRY_FORWARD_REASONS = CARRY_FORWARD_IDS;
|
|
87
79
|
/**
|
|
88
80
|
* Does this partial output carry ANSWER CONTENT, or is it the model clearing its
|
|
89
81
|
* throat?
|
|
90
82
|
*
|
|
91
|
-
*
|
|
92
|
-
*
|
|
93
|
-
* attempts and salvage shipped this as the section:
|
|
83
|
+
* Keeping the LONGEST partial is not the same question, and it has an obvious
|
|
84
|
+
* failure: a preamble sentence like
|
|
94
85
|
*
|
|
95
86
|
* "Now let me get more details on the specific APIs and components I need:"
|
|
96
87
|
*
|
|
97
|
-
*
|
|
98
|
-
* nothing. Both trials scored 2 entries and DEGRADED, against 22 and 5 for the
|
|
99
|
-
* same fixtures in baseline.
|
|
88
|
+
* beats an empty string on length and carries nothing.
|
|
100
89
|
*
|
|
101
90
|
* A research worker's answer is a list of lines that each name something and
|
|
102
|
-
* describe it. The test is therefore structural, not lexical: at least
|
|
103
|
-
* that look like entries — a name, then a gap, then a description. Prose wraps
|
|
104
|
-
*
|
|
91
|
+
* describe it. The test is therefore structural, not lexical: at least TWO lines
|
|
92
|
+
* that look like entries — a name, then a gap, then a description. Prose wraps at
|
|
93
|
+
* no particular column and does not repeat that shape; the sentence above scores
|
|
94
|
+
* zero entry lines.
|
|
105
95
|
*/
|
|
106
96
|
export function hasAnswerContent(text) {
|
|
107
97
|
return text.split('\n').filter(isEntryLine).length >= 2;
|
|
@@ -109,13 +99,15 @@ export function hasAnswerContent(text) {
|
|
|
109
99
|
/**
|
|
110
100
|
* Is ONE line an entry — a name, a gap, then a description — rather than prose?
|
|
111
101
|
*
|
|
112
|
-
* Split out of `hasAnswerContent` so the same rule can decide what a line IS,
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
*
|
|
102
|
+
* Split out of `hasAnswerContent` so the same rule can decide what a line IS, not
|
|
103
|
+
* just how many of them there are — a reader of a FILES section needs the same
|
|
104
|
+
* test, and its own idea of an entry would count a preamble sentence or a leaked
|
|
105
|
+
* `</tool_call>` as one.
|
|
116
106
|
*
|
|
117
|
-
*
|
|
118
|
-
* spaced dash
|
|
107
|
+
* A leading `-`, `*`, `•` or `1.`/`1)` bullet is stripped first. What remains must
|
|
108
|
+
* hold a two-space gap or a spaced dash and must NOT end in `.` or `:`. Prose
|
|
109
|
+
* wraps at no particular column, so it carries neither; when it does carry one, it
|
|
110
|
+
* ends in punctuation and an entry does not.
|
|
119
111
|
*/
|
|
120
112
|
export function isEntryLine(raw) {
|
|
121
113
|
const l = raw.replace(/^\s*(?:[-*•]|\d+[.)])\s+/, '').trim();
|
|
@@ -183,10 +175,10 @@ absoluteCeilingMs) {
|
|
|
183
175
|
return {
|
|
184
176
|
signal: ctrl.signal,
|
|
185
177
|
timedOut: () => timedOut,
|
|
186
|
-
//
|
|
187
|
-
//
|
|
188
|
-
//
|
|
189
|
-
//
|
|
178
|
+
// Push the deadline out, never past `started + ceilingMs`. Inert unless a
|
|
179
|
+
// caller calls it. A disabled timeout (nothing armed) stays disabled —
|
|
180
|
+
// extending "never" is meaningless — and an already-fired timer is not
|
|
181
|
+
// resurrected.
|
|
190
182
|
extend: (byMs, ceilingMs) => {
|
|
191
183
|
if (!armed || timedOut || ctrl.signal.aborted)
|
|
192
184
|
return;
|
|
@@ -232,9 +224,9 @@ absoluteCeilingMs) {
|
|
|
232
224
|
* commandTimeoutHint, which tells the model in as many words to bound its
|
|
233
225
|
* command; a SECOND hang means it ignored an explicit instruction, and a third
|
|
234
226
|
* means it ignored it twice. Giving a non-complying child the full ceiling again
|
|
235
|
-
*
|
|
236
|
-
*
|
|
237
|
-
*
|
|
227
|
+
* makes the worst case three times the ceiling, resting entirely on the model
|
|
228
|
+
* obeying prose. Halving bounds it at under twice the ceiling while costing a
|
|
229
|
+
* complying child nothing.
|
|
238
230
|
*
|
|
239
231
|
* `priorHangs` counts watchdog kills specifically, NOT total restarts — the
|
|
240
232
|
* restart budget is shared with loop kills, and a child restarted for LOOPING
|
|
@@ -242,8 +234,9 @@ absoluteCeilingMs) {
|
|
|
242
234
|
* the full ceiling. Only a hang after a hang is defiance.
|
|
243
235
|
*
|
|
244
236
|
* Floored at 30s so repeated halving cannot shrink the ceiling to something no
|
|
245
|
-
* real command could finish inside — but
|
|
246
|
-
*
|
|
237
|
+
* real command could finish inside — but the floor is `min(base, 30s)`, never
|
|
238
|
+
* above the configured ceiling, so a caller asking for 10s keeps 10s at every
|
|
239
|
+
* hang count. A base of 0 or less disables the watchdog and stays 0.
|
|
247
240
|
*/
|
|
248
241
|
export function commandCeilingForAttempt(baseMs, priorHangs) {
|
|
249
242
|
if (!(baseMs > 0))
|
|
@@ -323,8 +316,8 @@ export const RESTART_RULES = [
|
|
|
323
316
|
// a loop also tripped — the loop hint above is more specific.
|
|
324
317
|
reason: 'worker-timeout',
|
|
325
318
|
detect: s => s.timedOut && !s.loopHit && s.restartBudgetSpent < MAX_LOOP_RESTARTS ?
|
|
326
|
-
// The EFFECTIVE cap, which
|
|
327
|
-
// configured one would misname why this attempt died.
|
|
319
|
+
// The EFFECTIVE cap, which `extend`/`progress` can have moved —
|
|
320
|
+
// reporting the configured one would misname why this attempt died.
|
|
328
321
|
{ detail: `cap ${s.effectiveCapMs}ms` }
|
|
329
322
|
: null,
|
|
330
323
|
hint: () => WORKER_TIMEOUT_HINT,
|
|
@@ -332,22 +325,15 @@ export const RESTART_RULES = [
|
|
|
332
325
|
},
|
|
333
326
|
{
|
|
334
327
|
// A connection-class model error is restartable on the same budget, exactly
|
|
335
|
-
// as runPhaseChild already treats it
|
|
336
|
-
//
|
|
337
|
-
//
|
|
328
|
+
// as runPhaseChild already treats it. Without it one dropped socket fails
|
|
329
|
+
// the whole task at research, while the identical blip in refine or compose
|
|
330
|
+
// is absorbed.
|
|
338
331
|
//
|
|
339
|
-
//
|
|
340
|
-
//
|
|
341
|
-
//
|
|
342
|
-
//
|
|
343
|
-
//
|
|
344
|
-
// a re-spawn only helps when the outage outlasts it. It does: at a 20s
|
|
345
|
-
// outage the baseline never recovered and this policy always did, 0/8 → 8/8
|
|
346
|
-
// (Fisher p=0.00016), and the same at 35s. Below ~15s pi absorbs it alone —
|
|
347
|
-
// 8/8 both arms, so the retry neither helps nor costs there. Beyond ~46s
|
|
348
|
-
// (three spawns' combined budget) both arms fail. The price is paid only on
|
|
349
|
-
// a backend that is really gone: time-to-report goes ~15s → ~46s. Re-run:
|
|
350
|
-
// scripts/connection-retry-ab.ts.
|
|
332
|
+
// pi retries a failed turn itself before reporting anything, and a run that
|
|
333
|
+
// recovers reports no modelError at all (see JsonEventSink). So a SURFACED
|
|
334
|
+
// connection error means pi's own budget is already spent, and a re-spawn
|
|
335
|
+
// only helps when the outage outlasts it. The price is paid only on a
|
|
336
|
+
// backend that is really gone: time-to-report grows by the extra spawns.
|
|
351
337
|
//
|
|
352
338
|
// Connection class ONLY. Auth, bad request and context overflow still fail
|
|
353
339
|
// fast: re-issuing the same request cannot fix them, so spending the budget
|
|
@@ -380,11 +366,11 @@ export const RESTART_RULES = [
|
|
|
380
366
|
* runChild turns into a process-GROUP kill — reaping the hung command itself,
|
|
381
367
|
* not just the pi child holding it.
|
|
382
368
|
*
|
|
383
|
-
* LIMIT: the group kill only reaches processes still IN the group. A hung
|
|
384
|
-
*
|
|
385
|
-
* the fresh attempt can
|
|
386
|
-
*
|
|
387
|
-
*
|
|
369
|
+
* LIMIT: the group kill only reaches processes still IN the group. A hung command
|
|
370
|
+
* that detached a daemon (setsid, nohup, a background dev server) leaves it
|
|
371
|
+
* running, so the fresh attempt can hit a port the dead attempt's escapee still
|
|
372
|
+
* holds. There is no cheap fix from here; the restart hint's "check current state"
|
|
373
|
+
* line is the mitigation.
|
|
388
374
|
*
|
|
389
375
|
* Returns null when the watchdog is off, so the caller keeps the plain timeout
|
|
390
376
|
* signal and no per-call bookkeeping happens at all.
|
|
@@ -428,6 +414,13 @@ function commandWatch(timeoutMs) {
|
|
|
428
414
|
}
|
|
429
415
|
export async function runWorker(input) {
|
|
430
416
|
const tools = input.tools ?? DEFAULT_TOOLS;
|
|
417
|
+
// `--mode json` makes pi emit structured events as they happen instead of
|
|
418
|
+
// buffering the assistant text and flushing on exit. Its print-mode source
|
|
419
|
+
// shows both halves: under `json` a session subscriber writes every event to
|
|
420
|
+
// stdout as it arrives, while under `text` NOTHING is written until after the
|
|
421
|
+
// prompt resolves, when the last assistant message is printed once. That is
|
|
422
|
+
// what makes the wait/work split real — onFirstByte would otherwise fire
|
|
423
|
+
// moments before close and leave workMs at nearly zero.
|
|
431
424
|
const baseArgs = [
|
|
432
425
|
...childBaseArgs(input.extensions ?? []),
|
|
433
426
|
...(input.thinking ?? []),
|
|
@@ -475,11 +468,10 @@ export async function runWorker(input) {
|
|
|
475
468
|
const salvage = { text: null };
|
|
476
469
|
for (;;) {
|
|
477
470
|
const carried = salvage.text === null ? null : formatCarryForward(salvage.text);
|
|
478
|
-
// Announce the INJECTION, not just the restart.
|
|
479
|
-
//
|
|
480
|
-
//
|
|
481
|
-
//
|
|
482
|
-
// downstream of here can show it.
|
|
471
|
+
// Announce the INJECTION, not just the restart. The prompt goes to the child
|
|
472
|
+
// on stdin, so no log downstream of here can show it — without this hook
|
|
473
|
+
// "the carry reached the re-spawn" could only be inferred from the answer,
|
|
474
|
+
// which is inferring what a worker did from what it produced.
|
|
483
475
|
if (carried !== null) {
|
|
484
476
|
input.onCarryForward?.({
|
|
485
477
|
attempt: restarts.length + 1,
|
|
@@ -506,8 +498,8 @@ export async function runWorker(input) {
|
|
|
506
498
|
null
|
|
507
499
|
: new StallDetector(guards.loop.progress.limit, guards.loop.progress.churnFactor);
|
|
508
500
|
// Arm the churn rule BEFORE the first tool call. pi's stream carries no
|
|
509
|
-
// context event
|
|
510
|
-
//
|
|
501
|
+
// context event at all, so waiting for one leaves the rule permanently
|
|
502
|
+
// disarmed. The parent knows the window at spawn time.
|
|
511
503
|
stallDetector?.noteContext(contextWindowTokens(input.contextWindow));
|
|
512
504
|
// Capture the hit the detector reports (it also returns it to the unified
|
|
513
505
|
// runner, which kills the child on a hit). Without capturing it here the
|
|
@@ -546,8 +538,9 @@ export async function runWorker(input) {
|
|
|
546
538
|
// A tool call is the worker working. Inert unless the
|
|
547
539
|
// caller opted into a progress-based deadline.
|
|
548
540
|
timeout.progress();
|
|
549
|
-
//
|
|
550
|
-
//
|
|
541
|
+
// Naming ONE tool and ONE of its parameters here would put
|
|
542
|
+
// that knowledge in the generic runner, so it asks the
|
|
543
|
+
// tool's own row in WORKER_CHANNELS instead.
|
|
551
544
|
if (clock.fanout
|
|
552
545
|
&& workerChannel(call.name)?.isProjectSourceLookup?.(call.args ?? {}) === true) {
|
|
553
546
|
timeout.extend(clock.fanout.perLookupMs, clock.fanout.ceilingMs);
|
|
@@ -570,11 +563,11 @@ export async function runWorker(input) {
|
|
|
570
563
|
timeout.progress();
|
|
571
564
|
input.onLine?.(line);
|
|
572
565
|
},
|
|
573
|
-
// Always wired
|
|
574
|
-
//
|
|
575
|
-
//
|
|
576
|
-
//
|
|
577
|
-
//
|
|
566
|
+
// Always wired, never conditional on the command watchdog: the
|
|
567
|
+
// sink only emits a tool-execution-end if a handler exists, and
|
|
568
|
+
// a completed tool call is the clearest progress signal there
|
|
569
|
+
// is. Without it a worker whose tool calls all succeed would
|
|
570
|
+
// still look idle to the deadline.
|
|
578
571
|
onToolResult: r => {
|
|
579
572
|
timeout.progress();
|
|
580
573
|
cmdWatch?.onEnd(r.toolCallId);
|
|
@@ -637,7 +630,7 @@ export async function runWorker(input) {
|
|
|
637
630
|
// there would just mislabel the real failure.
|
|
638
631
|
const leaked = result.exitCode === 0 && !result.aborted ? detectLeakedToolCall(text) : null;
|
|
639
632
|
// THE RESTART LADDER. Precedence is RESTART_RULES' row order; this loop
|
|
640
|
-
// owns the ritual every rule
|
|
633
|
+
// owns the ritual every rule would otherwise repeat: budget, hint, counters,
|
|
641
634
|
// record-and-announce, backoff, re-spawn.
|
|
642
635
|
const state = {
|
|
643
636
|
...(loopHit ? { loopHit } : {}),
|
|
@@ -678,24 +671,22 @@ export async function runWorker(input) {
|
|
|
678
671
|
}
|
|
679
672
|
if (restarted)
|
|
680
673
|
continue;
|
|
681
|
-
// SALVAGE.
|
|
682
|
-
//
|
|
683
|
-
//
|
|
684
|
-
//
|
|
685
|
-
//
|
|
686
|
-
// with a worse one.
|
|
674
|
+
// SALVAGE. Returning the LAST attempt's text unconditionally makes a
|
|
675
|
+
// worker whose final attempt was killed early report nothing at all — even
|
|
676
|
+
// when a discarded attempt produced a usable answer that was still in hand
|
|
677
|
+
// at the moment it was thrown away. A restart budget is meant to buy more
|
|
678
|
+
// chances at an answer, not to overwrite a good attempt with a worse one.
|
|
687
679
|
//
|
|
688
680
|
// Gated on the final attempt having FAILED, not on it being shorter. A
|
|
689
681
|
// worker that finished cleanly has answered, and a short answer is a
|
|
690
682
|
// legitimate answer — length would let a long half-finished fragment
|
|
691
683
|
// override a concise correct one, which is the opposite of the fix.
|
|
692
|
-
// ASK THE LADDER — do not restate it.
|
|
693
|
-
//
|
|
694
|
-
//
|
|
695
|
-
//
|
|
696
|
-
//
|
|
697
|
-
//
|
|
698
|
-
// outcome the comment above forbids.
|
|
684
|
+
// ASK THE LADDER — do not restate it. Hand-writing this test restates the
|
|
685
|
+
// taxonomy `worker-failure.ts` owns, and a restatement drifts: drop
|
|
686
|
+
// `leakedToolCall` or a plain non-zero `exitCode` from it and an attempt
|
|
687
|
+
// that produced nothing usable counts as NOT failed, so salvage is skipped
|
|
688
|
+
// and a good earlier partial is overwritten — the outcome the comment above
|
|
689
|
+
// forbids.
|
|
699
690
|
//
|
|
700
691
|
// The two non-kill terms stay explicit because `worker-failure.ts`
|
|
701
692
|
// deliberately excludes them as CONSUMER policy: an empty answer and a
|