@mjasnikovs/pi-task 0.38.28 → 0.38.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config/config.d.ts +70 -70
- package/dist/config/config.js +26 -35
- package/dist/config/extension-list.d.ts +6 -5
- package/dist/config/extension-list.js +3 -2
- package/dist/config/reasoning-args.d.ts +9 -7
- package/dist/config/reasoning-args.js +12 -10
- package/dist/config/reasoning.d.ts +44 -105
- package/dist/config/reasoning.js +27 -704
- package/dist/config/register.d.ts +34 -48
- package/dist/config/register.js +41 -51
- package/dist/config/tool-list.d.ts +16 -16
- package/dist/config/tool-list.js +1 -1
- package/dist/remote/bridge.d.ts +19 -10
- package/dist/remote/bridge.js +3 -2
- package/dist/remote/broadcast.js +3 -1
- package/dist/remote/events.js +12 -11
- package/dist/remote/history.d.ts +1 -1
- package/dist/remote/protocol.d.ts +6 -3
- package/dist/remote/protocol.js +2 -1
- package/dist/remote/push.d.ts +16 -16
- package/dist/remote/push.js +27 -27
- package/dist/remote/register.d.ts +3 -3
- package/dist/remote/register.js +17 -19
- package/dist/remote/server.d.ts +9 -8
- package/dist/remote/server.js +15 -14
- package/dist/remote/session-state.d.ts +5 -4
- package/dist/remote/session-state.js +8 -5
- package/dist/remote/sw.d.ts +7 -6
- package/dist/remote/sw.js +7 -6
- package/dist/remote/tailscale.d.ts +4 -2
- package/dist/remote/tailscale.js +4 -2
- package/dist/remote/ui-highlight.js +6 -5
- package/dist/remote/ui-render.js +4 -4
- package/dist/remote/ui-script.js +24 -24
- package/dist/remote/ui-styles.d.ts +1 -1
- package/dist/remote/ui-styles.js +10 -13
- package/dist/remote/ui-tools.js +9 -6
- package/dist/shared/child-extensions.d.ts +29 -17
- package/dist/shared/child-extensions.js +29 -17
- package/dist/shared/child-output.d.ts +30 -24
- package/dist/shared/child-output.js +25 -17
- package/dist/shared/child-process.d.ts +47 -40
- package/dist/shared/child-process.js +50 -59
- package/dist/shared/command-watchdog.d.ts +22 -16
- package/dist/shared/command-watchdog.js +28 -21
- package/dist/shared/fs-text.d.ts +16 -10
- package/dist/shared/fs-text.js +16 -10
- package/dist/shared/git-runner.d.ts +25 -25
- package/dist/shared/git-runner.js +25 -25
- package/dist/shared/leaked-tool-call.d.ts +17 -11
- package/dist/shared/leaked-tool-call.js +23 -15
- package/dist/shared/model-endpoint.d.ts +29 -16
- package/dist/shared/model-endpoint.js +33 -21
- package/dist/shared/pi-invocation.d.ts +7 -4
- package/dist/shared/pi-invocation.js +12 -7
- package/dist/shared/pkg-version.d.ts +13 -5
- package/dist/shared/pkg-version.js +13 -5
- package/dist/shared/reasoning-capability.d.ts +35 -24
- package/dist/shared/reasoning-capability.js +35 -24
- package/dist/shared/stream-watchdog.d.ts +60 -44
- package/dist/shared/stream-watchdog.js +62 -45
- package/dist/task/accept-debt.d.ts +41 -43
- package/dist/task/accept-debt.js +73 -65
- package/dist/task/api-synthesis.d.ts +24 -21
- package/dist/task/api-synthesis.js +32 -26
- package/dist/task/apis-contract.d.ts +32 -64
- package/dist/task/apis-contract.js +32 -64
- package/dist/task/artifact-closure.d.ts +27 -13
- package/dist/task/artifact-closure.js +95 -67
- package/dist/task/auto-commit.d.ts +46 -35
- package/dist/task/auto-commit.js +51 -38
- package/dist/task/auto-io.d.ts +45 -25
- package/dist/task/auto-io.js +57 -29
- package/dist/task/auto-orchestrator.d.ts +26 -24
- package/dist/task/auto-orchestrator.js +178 -162
- package/dist/task/auto-prompts.d.ts +36 -24
- package/dist/task/auto-prompts.js +40 -26
- package/dist/task/autofix-ledger.d.ts +27 -25
- package/dist/task/autofix-ledger.js +29 -26
- package/dist/task/batch-test-task.d.ts +20 -12
- package/dist/task/batch-test-task.js +67 -60
- package/dist/task/boot-probe.d.ts +60 -44
- package/dist/task/boot-probe.js +91 -72
- package/dist/task/cancel-input.d.ts +30 -16
- package/dist/task/cancel-input.js +20 -11
- package/dist/task/cancel-points.d.ts +27 -20
- package/dist/task/cancel-points.js +30 -22
- package/dist/task/child-runner.d.ts +46 -51
- package/dist/task/child-runner.js +48 -49
- package/dist/task/child-status.d.ts +23 -16
- package/dist/task/child-status.js +23 -16
- package/dist/task/clamp-output.js +12 -5
- package/dist/task/command-run.d.ts +31 -28
- package/dist/task/command-run.js +44 -35
- package/dist/task/command-shrink.d.ts +25 -18
- package/dist/task/command-shrink.js +37 -31
- package/dist/task/command-watchdog.d.ts +9 -6
- package/dist/task/command-watchdog.js +21 -15
- package/dist/task/context-attribution.d.ts +34 -26
- package/dist/task/context-attribution.js +34 -26
- package/dist/task/context-silence.d.ts +39 -29
- package/dist/task/context-silence.js +35 -25
- package/dist/task/context-usage.d.ts +25 -7
- package/dist/task/context-usage.js +21 -6
- package/dist/task/contracts.d.ts +8 -4
- package/dist/task/contracts.js +25 -17
- package/dist/task/coverage-loop.d.ts +22 -18
- package/dist/task/coverage-loop.js +35 -30
- package/dist/task/critique-probes.d.ts +13 -14
- package/dist/task/critique-probes.js +50 -39
- package/dist/task/debug-log.d.ts +13 -5
- package/dist/task/debug-log.js +32 -20
- package/dist/task/decompose-fidelity.d.ts +11 -9
- package/dist/task/decompose-fidelity.js +38 -33
- package/dist/task/decompose-granularity.d.ts +41 -38
- package/dist/task/decompose-granularity.js +41 -38
- package/dist/task/deep-render-check.d.ts +22 -14
- package/dist/task/deep-render-check.js +40 -31
- package/dist/task/dropped-input.d.ts +12 -7
- package/dist/task/dropped-input.js +5 -2
- package/dist/task/enforce-attribution.d.ts +38 -47
- package/dist/task/enforce-attribution.js +46 -52
- package/dist/task/enforce-guidelines.d.ts +31 -20
- package/dist/task/enforce-guidelines.js +32 -21
- package/dist/task/enrichment.d.ts +7 -2
- package/dist/task/enrichment.js +26 -14
- package/dist/task/env-notes.d.ts +16 -7
- package/dist/task/env-notes.js +48 -31
- package/dist/task/env-template-closure.d.ts +4 -4
- package/dist/task/env-template-closure.js +42 -34
- package/dist/task/external-context.d.ts +28 -21
- package/dist/task/external-context.js +17 -12
- package/dist/task/failure-classifier.d.ts +4 -5
- package/dist/task/failure-classifier.js +6 -7
- package/dist/task/file-inventory.d.ts +15 -11
- package/dist/task/file-inventory.js +25 -22
- package/dist/task/final-gate-fix.d.ts +74 -86
- package/dist/task/final-gate-fix.js +97 -116
- package/dist/task/final-gate-progress.d.ts +29 -46
- package/dist/task/final-gate-progress.js +40 -51
- package/dist/task/final-gate.d.ts +64 -97
- package/dist/task/final-gate.js +192 -199
- package/dist/task/fix-child.d.ts +21 -27
- package/dist/task/fix-child.js +21 -27
- package/dist/task/foreign-path.d.ts +6 -5
- package/dist/task/foreign-path.js +0 -0
- package/dist/task/frozen-conflict.d.ts +9 -10
- package/dist/task/frozen-conflict.js +61 -64
- package/dist/task/frozen-path-guard.d.ts +35 -14
- package/dist/task/frozen-path-guard.js +56 -39
- package/dist/task/gate-child.d.ts +27 -28
- package/dist/task/gate-child.js +37 -35
- package/dist/task/gate-deps.d.ts +34 -27
- package/dist/task/gate-deps.js +169 -159
- package/dist/task/gate-tally.d.ts +77 -80
- package/dist/task/gate-tally.js +65 -68
- package/dist/task/git-state-guard.d.ts +15 -11
- package/dist/task/git-state-guard.js +76 -66
- package/dist/task/impl-widget.d.ts +25 -16
- package/dist/task/impl-widget.js +27 -17
- package/dist/task/implementation-thinking.d.ts +33 -31
- package/dist/task/implementation-thinking.js +5 -6
- package/dist/task/implementation-turn.d.ts +34 -31
- package/dist/task/implementation-turn.js +29 -27
- package/dist/task/inline-markdown.d.ts +20 -7
- package/dist/task/inline-markdown.js +15 -6
- package/dist/task/launch-config-gap.js +25 -39
- package/dist/task/launch-contract.d.ts +18 -21
- package/dist/task/launch-contract.js +28 -30
- package/dist/task/launch-manifest.d.ts +6 -2
- package/dist/task/launch-manifest.js +35 -34
- package/dist/task/ledger.js +16 -14
- package/dist/task/lint-fix.d.ts +6 -8
- package/dist/task/lint-fix.js +67 -69
- package/dist/task/loop-detector.d.ts +9 -8
- package/dist/task/loop-detector.js +16 -12
- package/dist/task/mid-run-input.d.ts +17 -15
- package/dist/task/mid-run-input.js +17 -15
- package/dist/task/orchestrator.d.ts +24 -28
- package/dist/task/orchestrator.js +62 -64
- package/dist/task/orientation.d.ts +18 -23
- package/dist/task/orientation.js +24 -31
- package/dist/task/owned-freeze-conflict.d.ts +21 -20
- package/dist/task/owned-freeze-conflict.js +52 -85
- package/dist/task/owned-freeze-reassign.d.ts +40 -60
- package/dist/task/owned-freeze-reassign.js +41 -61
- package/dist/task/parsers.d.ts +4 -2
- package/dist/task/parsers.js +4 -4
- package/dist/task/phases.d.ts +41 -48
- package/dist/task/phases.js +180 -248
- package/dist/task/plan-io.d.ts +6 -7
- package/dist/task/plan-io.js +6 -7
- package/dist/task/plan-orchestrator.d.ts +10 -8
- package/dist/task/plan-orchestrator.js +14 -10
- package/dist/task/plan-prompts.d.ts +6 -5
- package/dist/task/plan-prompts.js +6 -5
- package/dist/task/plan-readonly.d.ts +4 -5
- package/dist/task/plan-readonly.js +4 -5
- package/dist/task/plan-rounds.d.ts +17 -29
- package/dist/task/plan-rounds.js +21 -34
- package/dist/task/plan-session.d.ts +58 -72
- package/dist/task/plan-session.js +61 -83
- package/dist/task/probe-gaming.d.ts +28 -27
- package/dist/task/probe-gaming.js +0 -0
- package/dist/task/prohibition-probe.d.ts +14 -16
- package/dist/task/prompts.d.ts +3 -4
- package/dist/task/prompts.js +17 -26
- package/dist/task/qa-transcript.d.ts +15 -22
- package/dist/task/qa-transcript.js +15 -21
- package/dist/task/question-box.d.ts +17 -13
- package/dist/task/question-box.js +19 -15
- package/dist/task/question-dedup.d.ts +6 -7
- package/dist/task/question-dedup.js +13 -14
- package/dist/task/question-dialog.d.ts +22 -32
- package/dist/task/question-dialog.js +22 -32
- package/dist/task/question-source.d.ts +18 -44
- package/dist/task/question-source.js +22 -51
- package/dist/task/refuted-constraint.d.ts +11 -31
- package/dist/task/refuted-constraint.js +27 -51
- package/dist/task/regenerable-artifacts.d.ts +12 -31
- package/dist/task/regenerable-artifacts.js +12 -31
- package/dist/task/render-check.d.ts +11 -22
- package/dist/task/render-check.js +33 -46
- package/dist/task/repo-health-check.d.ts +10 -14
- package/dist/task/repo-health-check.js +17 -23
- package/dist/task/requirements.d.ts +38 -71
- package/dist/task/requirements.js +78 -126
- package/dist/task/research-fanout-budget.d.ts +51 -88
- package/dist/task/research-fanout-budget.js +51 -88
- package/dist/task/research-worker.d.ts +33 -36
- package/dist/task/research-worker.js +39 -61
- package/dist/task/resume-gap.d.ts +14 -15
- package/dist/task/root-cause-repair.d.ts +9 -9
- package/dist/task/root-cause-repair.js +28 -40
- package/dist/task/run-bracket.d.ts +10 -13
- package/dist/task/run-end.d.ts +12 -22
- package/dist/task/run-end.js +8 -16
- package/dist/task/run-final-gate.d.ts +19 -21
- package/dist/task/run-final-gate.js +62 -80
- package/dist/task/runner-globs.d.ts +12 -13
- package/dist/task/runner-globs.js +12 -13
- package/dist/task/runner-resolve.d.ts +9 -9
- package/dist/task/runner-resolve.js +22 -23
- package/dist/task/script-escape.d.ts +10 -12
- package/dist/task/script-escape.js +13 -14
- package/dist/task/serve-entry.d.ts +1 -1
- package/dist/task/serve-entry.js +22 -25
- package/dist/task/service-blocks.js +4 -2
- package/dist/task/shipped-source.d.ts +11 -29
- package/dist/task/shipped-source.js +11 -29
- package/dist/task/skip-escape.js +10 -14
- package/dist/task/spec-urls.d.ts +26 -65
- package/dist/task/spec-urls.js +26 -65
- package/dist/task/spec-validation.d.ts +17 -20
- package/dist/task/spec-validation.js +17 -20
- package/dist/task/stall-detector.d.ts +23 -30
- package/dist/task/stall-detector.js +23 -30
- package/dist/task/stream-watchdog.d.ts +14 -12
- package/dist/task/stream-watchdog.js +14 -12
- package/dist/task/substitution-probe.d.ts +17 -20
- package/dist/task/substitution-probe.js +17 -20
- package/dist/task/task-gates.d.ts +36 -41
- package/dist/task/task-gates.js +95 -106
- package/dist/task/task-io.d.ts +4 -4
- package/dist/task/task-io.js +4 -4
- package/dist/task/task-parsers.js +4 -3
- package/dist/task/task-provenance.d.ts +2 -2
- package/dist/task/task-provenance.js +11 -13
- package/dist/task/task-types.d.ts +4 -3
- package/dist/task/terminal-outcome.d.ts +14 -16
- package/dist/task/terminal-outcome.js +12 -14
- package/dist/task/test-assembly.d.ts +13 -20
- package/dist/task/test-assembly.js +13 -20
- package/dist/task/timings.d.ts +5 -3
- package/dist/task/timings.js +5 -3
- package/dist/task/title-label.d.ts +9 -4
- package/dist/task/title-label.js +9 -4
- package/dist/task/type-only-answer.d.ts +44 -52
- package/dist/task/type-only-answer.js +44 -52
- package/dist/task/unfailable-command.d.ts +18 -24
- package/dist/task/unfailable-command.js +21 -27
- package/dist/task/unknown-routing.d.ts +10 -4
- package/dist/task/unknown-routing.js +10 -4
- package/dist/task/user-directives.d.ts +5 -8
- package/dist/task/user-directives.js +5 -8
- package/dist/task/verify-quality.d.ts +18 -22
- package/dist/task/verify-quality.js +45 -46
- package/dist/task/verify-reconcile.d.ts +15 -10
- package/dist/task/verify-reconcile.js +45 -43
- package/dist/task/verify-resolution.d.ts +24 -20
- package/dist/task/verify-resolution.js +51 -50
- package/dist/task/verify-work.d.ts +59 -66
- package/dist/task/verify-work.js +101 -138
- package/dist/task/widget.d.ts +15 -14
- package/dist/task/widget.js +22 -17
- package/dist/task/wiring-claims.d.ts +25 -32
- package/dist/task/wiring-claims.js +30 -35
- package/dist/task/write-guard.d.ts +39 -39
- package/dist/task/write-guard.js +48 -51
- package/dist/task/yolo.d.ts +34 -30
- package/dist/task/yolo.js +42 -37
- package/dist/workers/abstention.d.ts +21 -41
- package/dist/workers/abstention.js +27 -48
- package/dist/workers/brave-search.d.ts +4 -3
- package/dist/workers/brave-search.js +5 -2
- package/dist/workers/brave-warning.d.ts +7 -4
- package/dist/workers/brave-warning.js +19 -7
- package/dist/workers/ddg-search.d.ts +6 -6
- package/dist/workers/ddg-search.js +18 -12
- package/dist/workers/docs-cache.js +5 -2
- package/dist/workers/docs-chunk.d.ts +30 -37
- package/dist/workers/docs-chunk.js +37 -41
- package/dist/workers/docs-core.d.ts +28 -44
- package/dist/workers/docs-core.js +25 -44
- package/dist/workers/docs-index.js +4 -3
- package/dist/workers/docs-lookup.d.ts +15 -22
- package/dist/workers/docs-lookup.js +12 -21
- package/dist/workers/docs-project.d.ts +15 -9
- package/dist/workers/docs-project.js +17 -10
- package/dist/workers/docs-resolve.d.ts +19 -20
- package/dist/workers/docs-resolve.js +35 -32
- package/dist/workers/docs-retrieve.d.ts +5 -6
- package/dist/workers/docs-retrieve.js +18 -15
- package/dist/workers/exa-search.d.ts +9 -6
- package/dist/workers/exa-search.js +23 -12
- package/dist/workers/fetch-core.d.ts +13 -16
- package/dist/workers/fetch-core.js +23 -23
- package/dist/workers/focused-extractor.d.ts +12 -12
- package/dist/workers/focused-extractor.js +16 -19
- package/dist/workers/html-clean.js +24 -14
- package/dist/workers/http-request.d.ts +28 -20
- package/dist/workers/http-request.js +22 -17
- package/dist/workers/npm-version.d.ts +28 -11
- package/dist/workers/npm-version.js +24 -15
- package/dist/workers/phantom-imports.d.ts +15 -12
- package/dist/workers/phantom-imports.js +30 -24
- package/dist/workers/pi-worker-core.d.ts +86 -54
- package/dist/workers/pi-worker-core.js +112 -112
- package/dist/workers/pi-worker-docs.d.ts +24 -19
- package/dist/workers/pi-worker-docs.js +67 -76
- package/dist/workers/pi-worker-fetch.d.ts +7 -3
- package/dist/workers/pi-worker-fetch.js +27 -19
- package/dist/workers/pi-worker-search.js +12 -8
- package/dist/workers/pi-worker.d.ts +9 -4
- package/dist/workers/pi-worker.js +23 -10
- package/dist/workers/reasoning-warning.d.ts +18 -17
- package/dist/workers/reasoning-warning.js +22 -20
- package/dist/workers/research-cache.js +50 -78
- package/dist/workers/search-core.js +7 -5
- package/dist/workers/search-types.d.ts +10 -9
- package/dist/workers/search-types.js +9 -8
- package/dist/workers/session-hint.d.ts +13 -14
- package/dist/workers/session-hint.js +8 -9
- package/dist/workers/shared.d.ts +21 -25
- package/dist/workers/shared.js +0 -0
- package/dist/workers/single-read-extension.d.ts +14 -7
- package/dist/workers/single-read-extension.js +14 -7
- package/dist/workers/single-read-guard.d.ts +25 -28
- package/dist/workers/single-read-guard.js +32 -32
- package/dist/workers/typeonly-log.d.ts +12 -9
- package/dist/workers/typeonly-log.js +29 -33
- package/dist/workers/worker-channels.d.ts +15 -23
- package/dist/workers/worker-channels.js +15 -23
- package/dist/workers/worker-failure.d.ts +38 -46
- package/dist/workers/worker-failure.js +31 -39
- package/dist/workers/worker-kill.d.ts +25 -26
- package/dist/workers/worker-kill.js +16 -19
- package/dist/workers/worker-profiles.d.ts +43 -53
- package/dist/workers/worker-profiles.js +30 -38
- package/package.json +10 -8
|
@@ -6,44 +6,48 @@ import { type WorkerGuardOverride, type WorkerGuardPolicy, type WorkerPolicyInpu
|
|
|
6
6
|
* command could be cited from. `pi-worker-docs` (the primary), `read` and `grep`
|
|
7
7
|
* (project source), and the web escalations `pi-worker-search`/`pi-worker-fetch`.
|
|
8
8
|
*
|
|
9
|
-
* `ls` and `find` are deliberately EXCLUDED: they return file
|
|
10
|
-
* and APIS owns symbols by name only, never paths (
|
|
11
|
-
* enumeration cannot verify a
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
9
|
+
* `ls` and `find` are deliberately EXCLUDED: they return file and directory
|
|
10
|
+
* NAMES, and APIS owns symbols by name only, never paths (see
|
|
11
|
+
* RESEARCH_APIS_PROMPT in prompts.ts). Bare enumeration cannot verify a
|
|
12
|
+
* signature, so a worker that fabricates its section from memory cannot launder
|
|
13
|
+
* itself grounded by calling `ls` once — "one trivial `ls` then fabricate the
|
|
14
|
+
* rest" still leaves groundingRetrievalCount at 0.
|
|
15
|
+
*
|
|
16
|
+
* The set is DERIVED from WORKER_CHANNELS (worker-channels.ts), not hand-kept.
|
|
17
|
+
* Re-exported here only so worker-channels.test.ts can assert this module hands
|
|
18
|
+
* back the same predicate.
|
|
15
19
|
*/
|
|
16
20
|
export { isGroundingRetrieval } from './worker-channels.js';
|
|
17
21
|
/**
|
|
18
22
|
* Does this partial output carry ANSWER CONTENT, or is it the model clearing its
|
|
19
23
|
* throat?
|
|
20
24
|
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
* attempts and salvage shipped this as the section:
|
|
25
|
+
* Keeping the LONGEST partial is not the same question, and it has an obvious
|
|
26
|
+
* failure: a preamble sentence like
|
|
24
27
|
*
|
|
25
28
|
* "Now let me get more details on the specific APIs and components I need:"
|
|
26
29
|
*
|
|
27
|
-
*
|
|
28
|
-
* nothing. Both trials scored 2 entries and DEGRADED, against 22 and 5 for the
|
|
29
|
-
* same fixtures in baseline.
|
|
30
|
+
* beats an empty string on length and carries nothing.
|
|
30
31
|
*
|
|
31
32
|
* A research worker's answer is a list of lines that each name something and
|
|
32
|
-
* describe it. The test is therefore structural, not lexical: at least
|
|
33
|
-
* that look like entries — a name, then a gap, then a description. Prose wraps
|
|
34
|
-
*
|
|
33
|
+
* describe it. The test is therefore structural, not lexical: at least TWO lines
|
|
34
|
+
* that look like entries — a name, then a gap, then a description. Prose wraps at
|
|
35
|
+
* no particular column and does not repeat that shape; the sentence above scores
|
|
36
|
+
* zero entry lines.
|
|
35
37
|
*/
|
|
36
38
|
export declare function hasAnswerContent(text: string): boolean;
|
|
37
39
|
/**
|
|
38
40
|
* Is ONE line an entry — a name, a gap, then a description — rather than prose?
|
|
39
41
|
*
|
|
40
|
-
* Split out of `hasAnswerContent` so the same rule can decide what a line IS,
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
42
|
+
* Split out of `hasAnswerContent` so the same rule can decide what a line IS, not
|
|
43
|
+
* just how many of them there are — a reader of a FILES section needs the same
|
|
44
|
+
* test, and its own idea of an entry would count a preamble sentence or a leaked
|
|
45
|
+
* `</tool_call>` as one.
|
|
44
46
|
*
|
|
45
|
-
*
|
|
46
|
-
* spaced dash
|
|
47
|
+
* A leading `-`, `*`, `•` or `1.`/`1)` bullet is stripped first. What remains must
|
|
48
|
+
* hold a two-space gap or a spaced dash and must NOT end in `.` or `:`. Prose
|
|
49
|
+
* wraps at no particular column, so it carries neither; when it does carry one, it
|
|
50
|
+
* ends in punctuation and an entry does not.
|
|
47
51
|
*/
|
|
48
52
|
export declare function isEntryLine(raw: string): boolean;
|
|
49
53
|
/**
|
|
@@ -67,7 +71,7 @@ export interface RunWorkerInput {
|
|
|
67
71
|
/** Called for each tool execution start and text-writing event inside the worker. */
|
|
68
72
|
onLine?: (line: string) => void;
|
|
69
73
|
/** Called when a tool call FINISHES, with its (truncatable) result — lets a caller
|
|
70
|
-
* log tool OUTPUTS, not just the command
|
|
74
|
+
* log tool OUTPUTS, not just the command. */
|
|
71
75
|
onToolResult?: (result: {
|
|
72
76
|
name: string;
|
|
73
77
|
isError: boolean;
|
|
@@ -82,20 +86,34 @@ export interface RunWorkerInput {
|
|
|
82
86
|
*/
|
|
83
87
|
onContextUsage?: (snapshot: ContextSnapshot) => void;
|
|
84
88
|
/**
|
|
85
|
-
* The worker child's context window in tokens
|
|
86
|
-
*
|
|
87
|
-
*
|
|
88
|
-
*
|
|
89
|
+
* The worker child's context window in tokens, or `'unknown'` when the
|
|
90
|
+
* caller genuinely has none.
|
|
91
|
+
*
|
|
92
|
+
* REQUIRED, and required for the same reason `profile` below is. pi's event
|
|
93
|
+
* stream carries NO window — the string `context_usage` appears nowhere in any
|
|
94
|
+
* installed @earendil-works package, and its only usage-bearing JSON event is
|
|
95
|
+
* `message_update` — so this parameter is the only source there is. Left
|
|
96
|
+
* optional, a caller that omits it leaves `noteContext` seeing 0, and
|
|
97
|
+
* `StallDetector`'s CONTEXT CHURN rule is gated on a positive window, so the
|
|
98
|
+
* rule silently does not exist. That reads exactly like a rule that exists and
|
|
99
|
+
* did not trip.
|
|
100
|
+
*
|
|
101
|
+
* WHY A WORD AND NOT `0` OR `null`. Both of those are what a caller types when
|
|
102
|
+
* it has not thought about the question, and both disarm the rule silently.
|
|
103
|
+
* `'unknown'` cannot be typed by accident, is greppable, and shows up in a diff
|
|
104
|
+
* as a decision.
|
|
105
|
+
*
|
|
106
|
+
* Two consumers read it: the churn rule, and the caller's progress bar, which
|
|
107
|
+
* shows a bare token count without a window. Both degrade on `'unknown'`.
|
|
89
108
|
*/
|
|
90
|
-
contextWindow
|
|
109
|
+
contextWindow: number | 'unknown';
|
|
91
110
|
/**
|
|
92
111
|
* WHICH KIND of worker child this is — the whole guard policy, in one word.
|
|
93
112
|
*
|
|
94
|
-
* REQUIRED, and required on purpose.
|
|
95
|
-
*
|
|
96
|
-
*
|
|
97
|
-
*
|
|
98
|
-
* children without anyone deciding it should. See worker-profiles.ts.
|
|
113
|
+
* REQUIRED, and required on purpose. As independent optionals, a caller that
|
|
114
|
+
* named none of the guard knobs still got a full policy and nobody could see
|
|
115
|
+
* which one — so a child can end up running the strictest wall clock of the
|
|
116
|
+
* three without anyone deciding it should. See worker-profiles.ts.
|
|
99
117
|
*/
|
|
100
118
|
profile: WorkerProfileId;
|
|
101
119
|
/**
|
|
@@ -104,10 +122,11 @@ export interface RunWorkerInput {
|
|
|
104
122
|
*/
|
|
105
123
|
policyInputs?: WorkerPolicyInputs;
|
|
106
124
|
/**
|
|
107
|
-
* Whole guard rows laid over the profile's. TESTS AND
|
|
108
|
-
*
|
|
109
|
-
*
|
|
110
|
-
*
|
|
125
|
+
* Whole guard rows laid over the profile's. TESTS AND HARNESSES ONLY — an
|
|
126
|
+
* override at a production call site is the hand-picked subset this design
|
|
127
|
+
* exists to stop. `worker-profiles.test.ts` enforces it: its "no production
|
|
128
|
+
* source file passes an `override` to runWorker" test scans src/ for a leading
|
|
129
|
+
* `override:` and fails on any hit.
|
|
111
130
|
*/
|
|
112
131
|
override?: WorkerGuardOverride;
|
|
113
132
|
/**
|
|
@@ -116,9 +135,9 @@ export interface RunWorkerInput {
|
|
|
116
135
|
*
|
|
117
136
|
* WHY: asserting that a profile RESOLVES correctly proves nothing about
|
|
118
137
|
* whether runWorker then READS it correctly — a rewiring that turns "0 means
|
|
119
|
-
* off" into "0 means on" leaves every profile assertion green. This hook
|
|
120
|
-
*
|
|
121
|
-
*
|
|
138
|
+
* off" into "0 means on" leaves every profile assertion green. This hook lets
|
|
139
|
+
* a test drive the REAL call site and read back the REAL policy instead of
|
|
140
|
+
* re-typing the table; worker-profiles.test.ts is where those assertions live.
|
|
122
141
|
*/
|
|
123
142
|
onPolicy?: (policy: WorkerGuardPolicy) => void;
|
|
124
143
|
/** Backoff sleep, injectable so tests don't wait out the real delays. */
|
|
@@ -151,12 +170,10 @@ export interface RunWorkerInput {
|
|
|
151
170
|
*
|
|
152
171
|
* WHY: every restart branch below throws away a whole attempt's wall clock
|
|
153
172
|
* along with its text, and `waitMs`/`workMs` describe the FINAL attempt only.
|
|
154
|
-
* With no hook here
|
|
155
|
-
*
|
|
156
|
-
*
|
|
157
|
-
*
|
|
158
|
-
* was only recoverable by subtracting reported wait+work from the timestamps
|
|
159
|
-
* of the `start` and `done` lines around it.
|
|
173
|
+
* With no hook here a discarded attempt is structurally invisible — the worker
|
|
174
|
+
* still returns `exitCode` 0 and reads as a clean success, and the lost time is
|
|
175
|
+
* recoverable only by subtracting the reported wait+work from the timestamps
|
|
176
|
+
* around the call.
|
|
160
177
|
*/
|
|
161
178
|
onRestart?: (restart: WorkerRestart) => void;
|
|
162
179
|
}
|
|
@@ -178,6 +195,19 @@ export interface WorkerRestart {
|
|
|
178
195
|
workMs: number;
|
|
179
196
|
/** Reason-specific diagnosis: the looping call, the hung tool, the error text. */
|
|
180
197
|
detail?: string;
|
|
198
|
+
/**
|
|
199
|
+
* Characters of ANSWER TEXT this attempt had produced at the moment it was
|
|
200
|
+
* thrown away.
|
|
201
|
+
*
|
|
202
|
+
* Recorded whatever `carryForward` says, because the DISCARD is the thing a
|
|
203
|
+
* reader cannot otherwise see. Without it, "the guards worked and the run
|
|
204
|
+
* returned almost nothing" and "the guards worked and the run threw away a
|
|
205
|
+
* finished answer" print identically.
|
|
206
|
+
*
|
|
207
|
+
* It is an OBSERVATION, not a decision: harvesting into `salvage` is still
|
|
208
|
+
* gated on the profile, and this number changes no behaviour.
|
|
209
|
+
*/
|
|
210
|
+
partialChars: number;
|
|
181
211
|
}
|
|
182
212
|
export interface RunWorkerResult {
|
|
183
213
|
text: string;
|
|
@@ -188,9 +218,9 @@ export interface RunWorkerResult {
|
|
|
188
218
|
* The provider-reported cause when the model turn itself failed (disconnect,
|
|
189
219
|
* fetch failed, 5xx after pi's own retries): pi delivers it as an assistant
|
|
190
220
|
* message with stopReason "error" and EMPTY text, exit code 0. Phase children
|
|
191
|
-
*
|
|
192
|
-
*
|
|
193
|
-
*
|
|
221
|
+
* surface this through child-runner.ts; without it a swallowed provider error
|
|
222
|
+
* reaches the caller as an indistinguishable empty answer and gets reported as
|
|
223
|
+
* the useless "produced no output".
|
|
194
224
|
* Only meaningful when `text` is empty: a turn that produced text after pi
|
|
195
225
|
* recovered is a success, and the first-error capture must not relabel it.
|
|
196
226
|
*/
|
|
@@ -315,9 +345,9 @@ export interface RunWorkerResult {
|
|
|
315
345
|
* commandTimeoutHint, which tells the model in as many words to bound its
|
|
316
346
|
* command; a SECOND hang means it ignored an explicit instruction, and a third
|
|
317
347
|
* means it ignored it twice. Giving a non-complying child the full ceiling again
|
|
318
|
-
*
|
|
319
|
-
*
|
|
320
|
-
*
|
|
348
|
+
* makes the worst case three times the ceiling, resting entirely on the model
|
|
349
|
+
* obeying prose. Halving bounds it at under twice the ceiling while costing a
|
|
350
|
+
* complying child nothing.
|
|
321
351
|
*
|
|
322
352
|
* `priorHangs` counts watchdog kills specifically, NOT total restarts — the
|
|
323
353
|
* restart budget is shared with loop kills, and a child restarted for LOOPING
|
|
@@ -325,8 +355,9 @@ export interface RunWorkerResult {
|
|
|
325
355
|
* the full ceiling. Only a hang after a hang is defiance.
|
|
326
356
|
*
|
|
327
357
|
* Floored at 30s so repeated halving cannot shrink the ceiling to something no
|
|
328
|
-
* real command could finish inside — but
|
|
329
|
-
*
|
|
358
|
+
* real command could finish inside — but the floor is `min(base, 30s)`, never
|
|
359
|
+
* above the configured ceiling, so a caller asking for 10s keeps 10s at every
|
|
360
|
+
* hang count. A base of 0 or less disables the watchdog and stays 0.
|
|
330
361
|
*/
|
|
331
362
|
export declare function commandCeilingForAttempt(baseMs: number, priorHangs: number): number;
|
|
332
363
|
/**
|
|
@@ -343,7 +374,8 @@ interface RestartState {
|
|
|
343
374
|
timedOut: boolean;
|
|
344
375
|
modelError?: string;
|
|
345
376
|
leaked: string | null;
|
|
346
|
-
/** The cap this attempt actually died against
|
|
377
|
+
/** The cap this attempt actually died against, not the configured one:
|
|
378
|
+
* `extend`/`progress` can push the deadline out during the attempt. */
|
|
347
379
|
effectiveCapMs: number;
|
|
348
380
|
/** The child's tool string, which decides whether its edits can persist. */
|
|
349
381
|
tools: string;
|
|
@@ -10,33 +10,36 @@ import { detectLeakedToolCall, leakedToolCallHint, MAX_LEAK_RETRIES } from '../s
|
|
|
10
10
|
import { discoverModelEndpoints, probeModelEndpoints } from '../shared/model-endpoint.js';
|
|
11
11
|
import { streamStallHint } from '../shared/stream-watchdog.js';
|
|
12
12
|
import { classifyWorkerFailure } from './worker-failure.js';
|
|
13
|
-
import { CARRY_FORWARD_IDS } from './worker-kill.js';
|
|
13
|
+
import { CARRY_FORWARD_IDS, RESTART_ORDER } from './worker-kill.js';
|
|
14
14
|
import { applyOverride, WORKER_PROFILES } from './worker-profiles.js';
|
|
15
|
-
|
|
16
|
-
// buffering the assistant text and flushing on exit. That matters for the
|
|
17
|
-
// wait/work timing split: in text mode the first stdout chunk only arrives at
|
|
18
|
-
// the very end, so onFirstByte fires moments before close and workMs is
|
|
19
|
-
// effectively zero. With JSON events the first byte lands as soon as the
|
|
20
|
-
// model starts producing — making waitMs the real queue/cold-start cost and
|
|
21
|
-
// workMs the real generation+tool-call cost.
|
|
15
|
+
/** The tool whitelist a caller gets when it names none. */
|
|
22
16
|
const DEFAULT_TOOLS = 'read,grep,find,ls';
|
|
17
|
+
/**
|
|
18
|
+
* The one place `'unknown'` becomes the 0 both consumers already treat as
|
|
19
|
+
* "no window". Written once so a future reader cannot re-introduce the optional
|
|
20
|
+
* by handling the union at only one of the two sites that read it.
|
|
21
|
+
*/
|
|
22
|
+
function contextWindowTokens(cw) {
|
|
23
|
+
return cw === 'unknown' || cw <= 0 ? 0 : cw;
|
|
24
|
+
}
|
|
23
25
|
/**
|
|
24
26
|
* Tool calls that can GROUND an APIS claim — i.e. return content a signature or
|
|
25
27
|
* command could be cited from. `pi-worker-docs` (the primary), `read` and `grep`
|
|
26
28
|
* (project source), and the web escalations `pi-worker-search`/`pi-worker-fetch`.
|
|
27
29
|
*
|
|
28
|
-
* `ls` and `find` are deliberately EXCLUDED: they return file
|
|
29
|
-
* and APIS owns symbols by name only, never paths (
|
|
30
|
-
* enumeration cannot verify a
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
30
|
+
* `ls` and `find` are deliberately EXCLUDED: they return file and directory
|
|
31
|
+
* NAMES, and APIS owns symbols by name only, never paths (see
|
|
32
|
+
* RESEARCH_APIS_PROMPT in prompts.ts). Bare enumeration cannot verify a
|
|
33
|
+
* signature, so a worker that fabricates its section from memory cannot launder
|
|
34
|
+
* itself grounded by calling `ls` once — "one trivial `ls` then fabricate the
|
|
35
|
+
* rest" still leaves groundingRetrievalCount at 0.
|
|
36
|
+
*
|
|
37
|
+
* The set is DERIVED from WORKER_CHANNELS (worker-channels.ts), not hand-kept.
|
|
38
|
+
* Re-exported here only so worker-channels.test.ts can assert this module hands
|
|
39
|
+
* back the same predicate.
|
|
34
40
|
*/
|
|
35
|
-
// The grounding set is derived from WORKER_CHANNELS (worker-channels.ts), not
|
|
36
|
-
// hand-kept — this was a second copy of the four tool names. Re-exported because
|
|
37
|
-
// several call sites and tests import it from here.
|
|
38
41
|
export { isGroundingRetrieval } from './worker-channels.js';
|
|
39
|
-
// RESEARCH_WORKER_TIMEOUT_MS and STALL_AFTER_MS live on the profile table
|
|
42
|
+
// RESEARCH_WORKER_TIMEOUT_MS and STALL_AFTER_MS live on the profile table
|
|
40
43
|
// (worker-profiles.ts): they are the default VALUES of two guard rows, and a
|
|
41
44
|
// default that lives apart from the table stating it is a second place to look.
|
|
42
45
|
/**
|
|
@@ -51,49 +54,44 @@ const WORKER_TIMEOUT_HINT = '[SYSTEM NOTE: Your previous attempt ran out of time
|
|
|
51
54
|
/**
|
|
52
55
|
* How much of a discarded attempt's answer is carried into the next one.
|
|
53
56
|
*
|
|
54
|
-
* A restart
|
|
55
|
-
*
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
* lookups, 5 of 5 workers burned the FULL restart budget, because every attempt
|
|
59
|
-
* re-read the same files against the same clock and died in the same place.
|
|
57
|
+
* A restart that hands the re-spawn nothing but a hint is why WORKER_TIMEOUT_HINT
|
|
58
|
+
* above can tell a worker "do not re-explore ground you have already covered"
|
|
59
|
+
* while giving it no record of what that ground was. It cannot comply, so it
|
|
60
|
+
* re-reads the same files against the same clock and dies in the same place.
|
|
60
61
|
*
|
|
61
|
-
* Carrying the partial
|
|
62
|
-
*
|
|
63
|
-
*
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
* framed as findings to VERIFY-or-DROP rather than as settled fact.
|
|
62
|
+
* Carrying the partial forward is what lets a restart converge instead of repeat.
|
|
63
|
+
* The risk is real: a half-written or speculative entry, replayed under "already
|
|
64
|
+
* established", is how a fabrication gets laundered into a final answer. That is
|
|
65
|
+
* why `formatCarryForward` frames it as findings to VERIFY-or-DROP rather than as
|
|
66
|
+
* settled fact.
|
|
67
67
|
*/
|
|
68
68
|
const CARRY_FORWARD_LIMIT = 24_000;
|
|
69
69
|
/**
|
|
70
70
|
* Restart reasons whose partial output is worth keeping.
|
|
71
71
|
*
|
|
72
|
-
*
|
|
73
|
-
*
|
|
74
|
-
*
|
|
75
|
-
*
|
|
76
|
-
*
|
|
72
|
+
* Exactly four, derived from WORKER_KILLS: `command-timeout`, `stream-stall`,
|
|
73
|
+
* `worker-timeout` and `connection-error` all discard work the model genuinely
|
|
74
|
+
* did. A loop kill and a leaked tool call do not — the first is by definition the
|
|
75
|
+
* same call repeated, the second is malformed protocol text, and replaying either
|
|
76
|
+
* would feed the failure back to itself.
|
|
77
77
|
*/
|
|
78
78
|
const CARRY_FORWARD_REASONS = CARRY_FORWARD_IDS;
|
|
79
79
|
/**
|
|
80
80
|
* Does this partial output carry ANSWER CONTENT, or is it the model clearing its
|
|
81
81
|
* throat?
|
|
82
82
|
*
|
|
83
|
-
*
|
|
84
|
-
*
|
|
85
|
-
* attempts and salvage shipped this as the section:
|
|
83
|
+
* Keeping the LONGEST partial is not the same question, and it has an obvious
|
|
84
|
+
* failure: a preamble sentence like
|
|
86
85
|
*
|
|
87
86
|
* "Now let me get more details on the specific APIs and components I need:"
|
|
88
87
|
*
|
|
89
|
-
*
|
|
90
|
-
* nothing. Both trials scored 2 entries and DEGRADED, against 22 and 5 for the
|
|
91
|
-
* same fixtures in baseline.
|
|
88
|
+
* beats an empty string on length and carries nothing.
|
|
92
89
|
*
|
|
93
90
|
* A research worker's answer is a list of lines that each name something and
|
|
94
|
-
* describe it. The test is therefore structural, not lexical: at least
|
|
95
|
-
* that look like entries — a name, then a gap, then a description. Prose wraps
|
|
96
|
-
*
|
|
91
|
+
* describe it. The test is therefore structural, not lexical: at least TWO lines
|
|
92
|
+
* that look like entries — a name, then a gap, then a description. Prose wraps at
|
|
93
|
+
* no particular column and does not repeat that shape; the sentence above scores
|
|
94
|
+
* zero entry lines.
|
|
97
95
|
*/
|
|
98
96
|
export function hasAnswerContent(text) {
|
|
99
97
|
return text.split('\n').filter(isEntryLine).length >= 2;
|
|
@@ -101,13 +99,15 @@ export function hasAnswerContent(text) {
|
|
|
101
99
|
/**
|
|
102
100
|
* Is ONE line an entry — a name, a gap, then a description — rather than prose?
|
|
103
101
|
*
|
|
104
|
-
* Split out of `hasAnswerContent` so the same rule can decide what a line IS,
|
|
105
|
-
*
|
|
106
|
-
*
|
|
107
|
-
*
|
|
102
|
+
* Split out of `hasAnswerContent` so the same rule can decide what a line IS, not
|
|
103
|
+
* just how many of them there are — a reader of a FILES section needs the same
|
|
104
|
+
* test, and its own idea of an entry would count a preamble sentence or a leaked
|
|
105
|
+
* `</tool_call>` as one.
|
|
108
106
|
*
|
|
109
|
-
*
|
|
110
|
-
* spaced dash
|
|
107
|
+
* A leading `-`, `*`, `•` or `1.`/`1)` bullet is stripped first. What remains must
|
|
108
|
+
* hold a two-space gap or a spaced dash and must NOT end in `.` or `:`. Prose
|
|
109
|
+
* wraps at no particular column, so it carries neither; when it does carry one, it
|
|
110
|
+
* ends in punctuation and an entry does not.
|
|
111
111
|
*/
|
|
112
112
|
export function isEntryLine(raw) {
|
|
113
113
|
const l = raw.replace(/^\s*(?:[-*•]|\d+[.)])\s+/, '').trim();
|
|
@@ -175,10 +175,10 @@ absoluteCeilingMs) {
|
|
|
175
175
|
return {
|
|
176
176
|
signal: ctrl.signal,
|
|
177
177
|
timedOut: () => timedOut,
|
|
178
|
-
//
|
|
179
|
-
//
|
|
180
|
-
//
|
|
181
|
-
//
|
|
178
|
+
// Push the deadline out, never past `started + ceilingMs`. Inert unless a
|
|
179
|
+
// caller calls it. A disabled timeout (nothing armed) stays disabled —
|
|
180
|
+
// extending "never" is meaningless — and an already-fired timer is not
|
|
181
|
+
// resurrected.
|
|
182
182
|
extend: (byMs, ceilingMs) => {
|
|
183
183
|
if (!armed || timedOut || ctrl.signal.aborted)
|
|
184
184
|
return;
|
|
@@ -224,9 +224,9 @@ absoluteCeilingMs) {
|
|
|
224
224
|
* commandTimeoutHint, which tells the model in as many words to bound its
|
|
225
225
|
* command; a SECOND hang means it ignored an explicit instruction, and a third
|
|
226
226
|
* means it ignored it twice. Giving a non-complying child the full ceiling again
|
|
227
|
-
*
|
|
228
|
-
*
|
|
229
|
-
*
|
|
227
|
+
* makes the worst case three times the ceiling, resting entirely on the model
|
|
228
|
+
* obeying prose. Halving bounds it at under twice the ceiling while costing a
|
|
229
|
+
* complying child nothing.
|
|
230
230
|
*
|
|
231
231
|
* `priorHangs` counts watchdog kills specifically, NOT total restarts — the
|
|
232
232
|
* restart budget is shared with loop kills, and a child restarted for LOOPING
|
|
@@ -234,8 +234,9 @@ absoluteCeilingMs) {
|
|
|
234
234
|
* the full ceiling. Only a hang after a hang is defiance.
|
|
235
235
|
*
|
|
236
236
|
* Floored at 30s so repeated halving cannot shrink the ceiling to something no
|
|
237
|
-
* real command could finish inside — but
|
|
238
|
-
*
|
|
237
|
+
* real command could finish inside — but the floor is `min(base, 30s)`, never
|
|
238
|
+
* above the configured ceiling, so a caller asking for 10s keeps 10s at every
|
|
239
|
+
* hang count. A base of 0 or less disables the watchdog and stays 0.
|
|
239
240
|
*/
|
|
240
241
|
export function commandCeilingForAttempt(baseMs, priorHangs) {
|
|
241
242
|
if (!(baseMs > 0))
|
|
@@ -315,8 +316,8 @@ export const RESTART_RULES = [
|
|
|
315
316
|
// a loop also tripped — the loop hint above is more specific.
|
|
316
317
|
reason: 'worker-timeout',
|
|
317
318
|
detect: s => s.timedOut && !s.loopHit && s.restartBudgetSpent < MAX_LOOP_RESTARTS ?
|
|
318
|
-
// The EFFECTIVE cap, which
|
|
319
|
-
// configured one would misname why this attempt died.
|
|
319
|
+
// The EFFECTIVE cap, which `extend`/`progress` can have moved —
|
|
320
|
+
// reporting the configured one would misname why this attempt died.
|
|
320
321
|
{ detail: `cap ${s.effectiveCapMs}ms` }
|
|
321
322
|
: null,
|
|
322
323
|
hint: () => WORKER_TIMEOUT_HINT,
|
|
@@ -324,22 +325,15 @@ export const RESTART_RULES = [
|
|
|
324
325
|
},
|
|
325
326
|
{
|
|
326
327
|
// A connection-class model error is restartable on the same budget, exactly
|
|
327
|
-
// as runPhaseChild already treats it
|
|
328
|
-
//
|
|
329
|
-
//
|
|
328
|
+
// as runPhaseChild already treats it. Without it one dropped socket fails
|
|
329
|
+
// the whole task at research, while the identical blip in refine or compose
|
|
330
|
+
// is absorbed.
|
|
330
331
|
//
|
|
331
|
-
//
|
|
332
|
-
//
|
|
333
|
-
//
|
|
334
|
-
//
|
|
335
|
-
//
|
|
336
|
-
// a re-spawn only helps when the outage outlasts it. It does: at a 20s
|
|
337
|
-
// outage the baseline never recovered and this policy always did, 0/8 → 8/8
|
|
338
|
-
// (Fisher p=0.00016), and the same at 35s. Below ~15s pi absorbs it alone —
|
|
339
|
-
// 8/8 both arms, so the retry neither helps nor costs there. Beyond ~46s
|
|
340
|
-
// (three spawns' combined budget) both arms fail. The price is paid only on
|
|
341
|
-
// a backend that is really gone: time-to-report goes ~15s → ~46s. Re-run:
|
|
342
|
-
// scripts/connection-retry-ab.ts.
|
|
332
|
+
// pi retries a failed turn itself before reporting anything, and a run that
|
|
333
|
+
// recovers reports no modelError at all (see JsonEventSink). So a SURFACED
|
|
334
|
+
// connection error means pi's own budget is already spent, and a re-spawn
|
|
335
|
+
// only helps when the outage outlasts it. The price is paid only on a
|
|
336
|
+
// backend that is really gone: time-to-report grows by the extra spawns.
|
|
343
337
|
//
|
|
344
338
|
// Connection class ONLY. Auth, bad request and context overflow still fail
|
|
345
339
|
// fast: re-issuing the same request cannot fix them, so spending the budget
|
|
@@ -372,11 +366,11 @@ export const RESTART_RULES = [
|
|
|
372
366
|
* runChild turns into a process-GROUP kill — reaping the hung command itself,
|
|
373
367
|
* not just the pi child holding it.
|
|
374
368
|
*
|
|
375
|
-
* LIMIT: the group kill only reaches processes still IN the group. A hung
|
|
376
|
-
*
|
|
377
|
-
* the fresh attempt can
|
|
378
|
-
*
|
|
379
|
-
*
|
|
369
|
+
* LIMIT: the group kill only reaches processes still IN the group. A hung command
|
|
370
|
+
* that detached a daemon (setsid, nohup, a background dev server) leaves it
|
|
371
|
+
* running, so the fresh attempt can hit a port the dead attempt's escapee still
|
|
372
|
+
* holds. There is no cheap fix from here; the restart hint's "check current state"
|
|
373
|
+
* line is the mitigation.
|
|
380
374
|
*
|
|
381
375
|
* Returns null when the watchdog is off, so the caller keeps the plain timeout
|
|
382
376
|
* signal and no per-call bookkeeping happens at all.
|
|
@@ -420,6 +414,13 @@ function commandWatch(timeoutMs) {
|
|
|
420
414
|
}
|
|
421
415
|
export async function runWorker(input) {
|
|
422
416
|
const tools = input.tools ?? DEFAULT_TOOLS;
|
|
417
|
+
// `--mode json` makes pi emit structured events as they happen instead of
|
|
418
|
+
// buffering the assistant text and flushing on exit. Its print-mode source
|
|
419
|
+
// shows both halves: under `json` a session subscriber writes every event to
|
|
420
|
+
// stdout as it arrives, while under `text` NOTHING is written until after the
|
|
421
|
+
// prompt resolves, when the last assistant message is printed once. That is
|
|
422
|
+
// what makes the wait/work split real — onFirstByte would otherwise fire
|
|
423
|
+
// moments before close and leave workMs at nearly zero.
|
|
423
424
|
const baseArgs = [
|
|
424
425
|
...childBaseArgs(input.extensions ?? []),
|
|
425
426
|
...(input.thinking ?? []),
|
|
@@ -467,11 +468,10 @@ export async function runWorker(input) {
|
|
|
467
468
|
const salvage = { text: null };
|
|
468
469
|
for (;;) {
|
|
469
470
|
const carried = salvage.text === null ? null : formatCarryForward(salvage.text);
|
|
470
|
-
// Announce the INJECTION, not just the restart.
|
|
471
|
-
//
|
|
472
|
-
//
|
|
473
|
-
//
|
|
474
|
-
// downstream of here can show it.
|
|
471
|
+
// Announce the INJECTION, not just the restart. The prompt goes to the child
|
|
472
|
+
// on stdin, so no log downstream of here can show it — without this hook
|
|
473
|
+
// "the carry reached the re-spawn" could only be inferred from the answer,
|
|
474
|
+
// which is inferring what a worker did from what it produced.
|
|
475
475
|
if (carried !== null) {
|
|
476
476
|
input.onCarryForward?.({
|
|
477
477
|
attempt: restarts.length + 1,
|
|
@@ -498,9 +498,9 @@ export async function runWorker(input) {
|
|
|
498
498
|
null
|
|
499
499
|
: new StallDetector(guards.loop.progress.limit, guards.loop.progress.churnFactor);
|
|
500
500
|
// Arm the churn rule BEFORE the first tool call. pi's stream carries no
|
|
501
|
-
// context event
|
|
502
|
-
//
|
|
503
|
-
stallDetector?.noteContext(input.contextWindow
|
|
501
|
+
// context event at all, so waiting for one leaves the rule permanently
|
|
502
|
+
// disarmed. The parent knows the window at spawn time.
|
|
503
|
+
stallDetector?.noteContext(contextWindowTokens(input.contextWindow));
|
|
504
504
|
// Capture the hit the detector reports (it also returns it to the unified
|
|
505
505
|
// runner, which kills the child on a hit). Without capturing it here the
|
|
506
506
|
// SIGTERM that kill produces would surface as a bare non-zero exit the
|
|
@@ -538,8 +538,9 @@ export async function runWorker(input) {
|
|
|
538
538
|
// A tool call is the worker working. Inert unless the
|
|
539
539
|
// caller opted into a progress-based deadline.
|
|
540
540
|
timeout.progress();
|
|
541
|
-
//
|
|
542
|
-
//
|
|
541
|
+
// Naming ONE tool and ONE of its parameters here would put
|
|
542
|
+
// that knowledge in the generic runner, so it asks the
|
|
543
|
+
// tool's own row in WORKER_CHANNELS instead.
|
|
543
544
|
if (clock.fanout
|
|
544
545
|
&& workerChannel(call.name)?.isProjectSourceLookup?.(call.args ?? {}) === true) {
|
|
545
546
|
timeout.extend(clock.fanout.perLookupMs, clock.fanout.ceilingMs);
|
|
@@ -562,11 +563,11 @@ export async function runWorker(input) {
|
|
|
562
563
|
timeout.progress();
|
|
563
564
|
input.onLine?.(line);
|
|
564
565
|
},
|
|
565
|
-
// Always wired
|
|
566
|
-
//
|
|
567
|
-
//
|
|
568
|
-
//
|
|
569
|
-
//
|
|
566
|
+
// Always wired, never conditional on the command watchdog: the
|
|
567
|
+
// sink only emits a tool-execution-end if a handler exists, and
|
|
568
|
+
// a completed tool call is the clearest progress signal there
|
|
569
|
+
// is. Without it a worker whose tool calls all succeed would
|
|
570
|
+
// still look idle to the deadline.
|
|
570
571
|
onToolResult: r => {
|
|
571
572
|
timeout.progress();
|
|
572
573
|
cmdWatch?.onEnd(r.toolCallId);
|
|
@@ -579,8 +580,8 @@ export async function runWorker(input) {
|
|
|
579
580
|
stallDetector?.noteContext(snapshot.contextWindow);
|
|
580
581
|
input.onContextUsage?.(snapshot);
|
|
581
582
|
},
|
|
582
|
-
...(input.contextWindow
|
|
583
|
-
{ contextWindow: input.contextWindow }
|
|
583
|
+
...(contextWindowTokens(input.contextWindow) > 0 ?
|
|
584
|
+
{ contextWindow: contextWindowTokens(input.contextWindow) }
|
|
584
585
|
: {})
|
|
585
586
|
}, input.spawn);
|
|
586
587
|
}
|
|
@@ -601,6 +602,7 @@ export async function runWorker(input) {
|
|
|
601
602
|
wallMs: tEnd - tAttemptStart,
|
|
602
603
|
waitMs,
|
|
603
604
|
workMs,
|
|
605
|
+
partialChars: text.trim().length,
|
|
604
606
|
...(detail ? { detail } : {})
|
|
605
607
|
};
|
|
606
608
|
restarts.push(record);
|
|
@@ -628,7 +630,7 @@ export async function runWorker(input) {
|
|
|
628
630
|
// there would just mislabel the real failure.
|
|
629
631
|
const leaked = result.exitCode === 0 && !result.aborted ? detectLeakedToolCall(text) : null;
|
|
630
632
|
// THE RESTART LADDER. Precedence is RESTART_RULES' row order; this loop
|
|
631
|
-
// owns the ritual every rule
|
|
633
|
+
// owns the ritual every rule would otherwise repeat: budget, hint, counters,
|
|
632
634
|
// record-and-announce, backoff, re-spawn.
|
|
633
635
|
const state = {
|
|
634
636
|
...(loopHit ? { loopHit } : {}),
|
|
@@ -669,24 +671,22 @@ export async function runWorker(input) {
|
|
|
669
671
|
}
|
|
670
672
|
if (restarted)
|
|
671
673
|
continue;
|
|
672
|
-
// SALVAGE.
|
|
673
|
-
//
|
|
674
|
-
//
|
|
675
|
-
//
|
|
676
|
-
//
|
|
677
|
-
// with a worse one.
|
|
674
|
+
// SALVAGE. Returning the LAST attempt's text unconditionally makes a
|
|
675
|
+
// worker whose final attempt was killed early report nothing at all — even
|
|
676
|
+
// when a discarded attempt produced a usable answer that was still in hand
|
|
677
|
+
// at the moment it was thrown away. A restart budget is meant to buy more
|
|
678
|
+
// chances at an answer, not to overwrite a good attempt with a worse one.
|
|
678
679
|
//
|
|
679
680
|
// Gated on the final attempt having FAILED, not on it being shorter. A
|
|
680
681
|
// worker that finished cleanly has answered, and a short answer is a
|
|
681
682
|
// legitimate answer — length would let a long half-finished fragment
|
|
682
683
|
// override a concise correct one, which is the opposite of the fix.
|
|
683
|
-
// ASK THE LADDER — do not restate it.
|
|
684
|
-
//
|
|
685
|
-
//
|
|
686
|
-
//
|
|
687
|
-
//
|
|
688
|
-
//
|
|
689
|
-
// outcome the comment above forbids.
|
|
684
|
+
// ASK THE LADDER — do not restate it. Hand-writing this test restates the
|
|
685
|
+
// taxonomy `worker-failure.ts` owns, and a restatement drifts: drop
|
|
686
|
+
// `leakedToolCall` or a plain non-zero `exitCode` from it and an attempt
|
|
687
|
+
// that produced nothing usable counts as NOT failed, so salvage is skipped
|
|
688
|
+
// and a good earlier partial is overwritten — the outcome the comment above
|
|
689
|
+
// forbids.
|
|
690
690
|
//
|
|
691
691
|
// The two non-kill terms stay explicit because `worker-failure.ts`
|
|
692
692
|
// deliberately excludes them as CONSUMER policy: an empty answer and a
|