@mjasnikovs/pi-task 0.38.28 → 0.38.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config/config.d.ts +70 -70
- package/dist/config/config.js +26 -35
- package/dist/config/extension-list.d.ts +6 -5
- package/dist/config/extension-list.js +3 -2
- package/dist/config/reasoning-args.d.ts +9 -7
- package/dist/config/reasoning-args.js +12 -10
- package/dist/config/reasoning.d.ts +44 -105
- package/dist/config/reasoning.js +27 -704
- package/dist/config/register.d.ts +34 -48
- package/dist/config/register.js +41 -51
- package/dist/config/tool-list.d.ts +16 -16
- package/dist/config/tool-list.js +1 -1
- package/dist/remote/bridge.d.ts +19 -10
- package/dist/remote/bridge.js +3 -2
- package/dist/remote/broadcast.js +3 -1
- package/dist/remote/events.js +12 -11
- package/dist/remote/history.d.ts +1 -1
- package/dist/remote/protocol.d.ts +6 -3
- package/dist/remote/protocol.js +2 -1
- package/dist/remote/push.d.ts +16 -16
- package/dist/remote/push.js +27 -27
- package/dist/remote/register.d.ts +3 -3
- package/dist/remote/register.js +17 -19
- package/dist/remote/server.d.ts +9 -8
- package/dist/remote/server.js +15 -14
- package/dist/remote/session-state.d.ts +5 -4
- package/dist/remote/session-state.js +8 -5
- package/dist/remote/sw.d.ts +7 -6
- package/dist/remote/sw.js +7 -6
- package/dist/remote/tailscale.d.ts +4 -2
- package/dist/remote/tailscale.js +4 -2
- package/dist/remote/ui-highlight.js +6 -5
- package/dist/remote/ui-render.js +4 -4
- package/dist/remote/ui-script.js +24 -24
- package/dist/remote/ui-styles.d.ts +1 -1
- package/dist/remote/ui-styles.js +10 -13
- package/dist/remote/ui-tools.js +9 -6
- package/dist/shared/child-extensions.d.ts +29 -17
- package/dist/shared/child-extensions.js +29 -17
- package/dist/shared/child-output.d.ts +30 -24
- package/dist/shared/child-output.js +25 -17
- package/dist/shared/child-process.d.ts +47 -40
- package/dist/shared/child-process.js +50 -59
- package/dist/shared/command-watchdog.d.ts +22 -16
- package/dist/shared/command-watchdog.js +28 -21
- package/dist/shared/fs-text.d.ts +16 -10
- package/dist/shared/fs-text.js +16 -10
- package/dist/shared/git-runner.d.ts +25 -25
- package/dist/shared/git-runner.js +25 -25
- package/dist/shared/leaked-tool-call.d.ts +17 -11
- package/dist/shared/leaked-tool-call.js +23 -15
- package/dist/shared/model-endpoint.d.ts +29 -16
- package/dist/shared/model-endpoint.js +33 -21
- package/dist/shared/pi-invocation.d.ts +7 -4
- package/dist/shared/pi-invocation.js +12 -7
- package/dist/shared/pkg-version.d.ts +13 -5
- package/dist/shared/pkg-version.js +13 -5
- package/dist/shared/reasoning-capability.d.ts +35 -24
- package/dist/shared/reasoning-capability.js +35 -24
- package/dist/shared/stream-watchdog.d.ts +60 -44
- package/dist/shared/stream-watchdog.js +62 -45
- package/dist/task/accept-debt.d.ts +41 -43
- package/dist/task/accept-debt.js +73 -65
- package/dist/task/api-synthesis.d.ts +24 -21
- package/dist/task/api-synthesis.js +32 -26
- package/dist/task/apis-contract.d.ts +32 -64
- package/dist/task/apis-contract.js +32 -64
- package/dist/task/artifact-closure.d.ts +27 -13
- package/dist/task/artifact-closure.js +95 -67
- package/dist/task/auto-commit.d.ts +46 -35
- package/dist/task/auto-commit.js +51 -38
- package/dist/task/auto-io.d.ts +45 -25
- package/dist/task/auto-io.js +57 -29
- package/dist/task/auto-orchestrator.d.ts +26 -24
- package/dist/task/auto-orchestrator.js +178 -162
- package/dist/task/auto-prompts.d.ts +36 -24
- package/dist/task/auto-prompts.js +40 -26
- package/dist/task/autofix-ledger.d.ts +27 -25
- package/dist/task/autofix-ledger.js +29 -26
- package/dist/task/batch-test-task.d.ts +20 -12
- package/dist/task/batch-test-task.js +67 -60
- package/dist/task/boot-probe.d.ts +60 -44
- package/dist/task/boot-probe.js +91 -72
- package/dist/task/cancel-input.d.ts +30 -16
- package/dist/task/cancel-input.js +20 -11
- package/dist/task/cancel-points.d.ts +27 -20
- package/dist/task/cancel-points.js +30 -22
- package/dist/task/child-runner.d.ts +46 -51
- package/dist/task/child-runner.js +48 -49
- package/dist/task/child-status.d.ts +23 -16
- package/dist/task/child-status.js +23 -16
- package/dist/task/clamp-output.js +12 -5
- package/dist/task/command-run.d.ts +31 -28
- package/dist/task/command-run.js +44 -35
- package/dist/task/command-shrink.d.ts +25 -18
- package/dist/task/command-shrink.js +37 -31
- package/dist/task/command-watchdog.d.ts +9 -6
- package/dist/task/command-watchdog.js +21 -15
- package/dist/task/context-attribution.d.ts +34 -26
- package/dist/task/context-attribution.js +34 -26
- package/dist/task/context-silence.d.ts +39 -29
- package/dist/task/context-silence.js +35 -25
- package/dist/task/context-usage.d.ts +25 -7
- package/dist/task/context-usage.js +21 -6
- package/dist/task/contracts.d.ts +8 -4
- package/dist/task/contracts.js +25 -17
- package/dist/task/coverage-loop.d.ts +22 -18
- package/dist/task/coverage-loop.js +35 -30
- package/dist/task/critique-probes.d.ts +13 -14
- package/dist/task/critique-probes.js +50 -39
- package/dist/task/debug-log.d.ts +13 -5
- package/dist/task/debug-log.js +32 -20
- package/dist/task/decompose-fidelity.d.ts +11 -9
- package/dist/task/decompose-fidelity.js +38 -33
- package/dist/task/decompose-granularity.d.ts +41 -38
- package/dist/task/decompose-granularity.js +41 -38
- package/dist/task/deep-render-check.d.ts +22 -14
- package/dist/task/deep-render-check.js +40 -31
- package/dist/task/dropped-input.d.ts +12 -7
- package/dist/task/dropped-input.js +5 -2
- package/dist/task/enforce-attribution.d.ts +38 -47
- package/dist/task/enforce-attribution.js +46 -52
- package/dist/task/enforce-guidelines.d.ts +31 -20
- package/dist/task/enforce-guidelines.js +32 -21
- package/dist/task/enrichment.d.ts +7 -2
- package/dist/task/enrichment.js +26 -14
- package/dist/task/env-notes.d.ts +16 -7
- package/dist/task/env-notes.js +48 -31
- package/dist/task/env-template-closure.d.ts +4 -4
- package/dist/task/env-template-closure.js +42 -34
- package/dist/task/external-context.d.ts +28 -21
- package/dist/task/external-context.js +17 -12
- package/dist/task/failure-classifier.d.ts +4 -5
- package/dist/task/failure-classifier.js +6 -7
- package/dist/task/file-inventory.d.ts +15 -11
- package/dist/task/file-inventory.js +25 -22
- package/dist/task/final-gate-fix.d.ts +74 -86
- package/dist/task/final-gate-fix.js +97 -116
- package/dist/task/final-gate-progress.d.ts +29 -46
- package/dist/task/final-gate-progress.js +40 -51
- package/dist/task/final-gate.d.ts +64 -97
- package/dist/task/final-gate.js +192 -199
- package/dist/task/fix-child.d.ts +21 -27
- package/dist/task/fix-child.js +21 -27
- package/dist/task/foreign-path.d.ts +6 -5
- package/dist/task/foreign-path.js +0 -0
- package/dist/task/frozen-conflict.d.ts +9 -10
- package/dist/task/frozen-conflict.js +61 -64
- package/dist/task/frozen-path-guard.d.ts +35 -14
- package/dist/task/frozen-path-guard.js +56 -39
- package/dist/task/gate-child.d.ts +27 -28
- package/dist/task/gate-child.js +37 -35
- package/dist/task/gate-deps.d.ts +34 -27
- package/dist/task/gate-deps.js +169 -159
- package/dist/task/gate-tally.d.ts +77 -80
- package/dist/task/gate-tally.js +65 -68
- package/dist/task/git-state-guard.d.ts +15 -11
- package/dist/task/git-state-guard.js +76 -66
- package/dist/task/impl-widget.d.ts +25 -16
- package/dist/task/impl-widget.js +27 -17
- package/dist/task/implementation-thinking.d.ts +33 -31
- package/dist/task/implementation-thinking.js +5 -6
- package/dist/task/implementation-turn.d.ts +34 -31
- package/dist/task/implementation-turn.js +29 -27
- package/dist/task/inline-markdown.d.ts +20 -7
- package/dist/task/inline-markdown.js +15 -6
- package/dist/task/launch-config-gap.js +25 -39
- package/dist/task/launch-contract.d.ts +18 -21
- package/dist/task/launch-contract.js +28 -30
- package/dist/task/launch-manifest.d.ts +6 -2
- package/dist/task/launch-manifest.js +35 -34
- package/dist/task/ledger.js +16 -14
- package/dist/task/lint-fix.d.ts +6 -8
- package/dist/task/lint-fix.js +67 -69
- package/dist/task/loop-detector.d.ts +9 -8
- package/dist/task/loop-detector.js +16 -12
- package/dist/task/mid-run-input.d.ts +17 -15
- package/dist/task/mid-run-input.js +17 -15
- package/dist/task/orchestrator.d.ts +24 -28
- package/dist/task/orchestrator.js +62 -64
- package/dist/task/orientation.d.ts +18 -23
- package/dist/task/orientation.js +24 -31
- package/dist/task/owned-freeze-conflict.d.ts +21 -20
- package/dist/task/owned-freeze-conflict.js +52 -85
- package/dist/task/owned-freeze-reassign.d.ts +40 -60
- package/dist/task/owned-freeze-reassign.js +41 -61
- package/dist/task/parsers.d.ts +4 -2
- package/dist/task/parsers.js +4 -4
- package/dist/task/phases.d.ts +41 -48
- package/dist/task/phases.js +180 -248
- package/dist/task/plan-io.d.ts +6 -7
- package/dist/task/plan-io.js +6 -7
- package/dist/task/plan-orchestrator.d.ts +10 -8
- package/dist/task/plan-orchestrator.js +14 -10
- package/dist/task/plan-prompts.d.ts +6 -5
- package/dist/task/plan-prompts.js +6 -5
- package/dist/task/plan-readonly.d.ts +4 -5
- package/dist/task/plan-readonly.js +4 -5
- package/dist/task/plan-rounds.d.ts +17 -29
- package/dist/task/plan-rounds.js +21 -34
- package/dist/task/plan-session.d.ts +58 -72
- package/dist/task/plan-session.js +61 -83
- package/dist/task/probe-gaming.d.ts +28 -27
- package/dist/task/probe-gaming.js +0 -0
- package/dist/task/prohibition-probe.d.ts +14 -16
- package/dist/task/prompts.d.ts +3 -4
- package/dist/task/prompts.js +17 -26
- package/dist/task/qa-transcript.d.ts +15 -22
- package/dist/task/qa-transcript.js +15 -21
- package/dist/task/question-box.d.ts +17 -13
- package/dist/task/question-box.js +19 -15
- package/dist/task/question-dedup.d.ts +6 -7
- package/dist/task/question-dedup.js +13 -14
- package/dist/task/question-dialog.d.ts +22 -32
- package/dist/task/question-dialog.js +22 -32
- package/dist/task/question-source.d.ts +18 -44
- package/dist/task/question-source.js +22 -51
- package/dist/task/refuted-constraint.d.ts +11 -31
- package/dist/task/refuted-constraint.js +27 -51
- package/dist/task/regenerable-artifacts.d.ts +12 -31
- package/dist/task/regenerable-artifacts.js +12 -31
- package/dist/task/render-check.d.ts +11 -22
- package/dist/task/render-check.js +33 -46
- package/dist/task/repo-health-check.d.ts +10 -14
- package/dist/task/repo-health-check.js +17 -23
- package/dist/task/requirements.d.ts +38 -71
- package/dist/task/requirements.js +78 -126
- package/dist/task/research-fanout-budget.d.ts +51 -88
- package/dist/task/research-fanout-budget.js +51 -88
- package/dist/task/research-worker.d.ts +33 -36
- package/dist/task/research-worker.js +39 -61
- package/dist/task/resume-gap.d.ts +14 -15
- package/dist/task/root-cause-repair.d.ts +9 -9
- package/dist/task/root-cause-repair.js +28 -40
- package/dist/task/run-bracket.d.ts +10 -13
- package/dist/task/run-end.d.ts +12 -22
- package/dist/task/run-end.js +8 -16
- package/dist/task/run-final-gate.d.ts +19 -21
- package/dist/task/run-final-gate.js +62 -80
- package/dist/task/runner-globs.d.ts +12 -13
- package/dist/task/runner-globs.js +12 -13
- package/dist/task/runner-resolve.d.ts +9 -9
- package/dist/task/runner-resolve.js +22 -23
- package/dist/task/script-escape.d.ts +10 -12
- package/dist/task/script-escape.js +13 -14
- package/dist/task/serve-entry.d.ts +1 -1
- package/dist/task/serve-entry.js +22 -25
- package/dist/task/service-blocks.js +4 -2
- package/dist/task/shipped-source.d.ts +11 -29
- package/dist/task/shipped-source.js +11 -29
- package/dist/task/skip-escape.js +10 -14
- package/dist/task/spec-urls.d.ts +26 -65
- package/dist/task/spec-urls.js +26 -65
- package/dist/task/spec-validation.d.ts +17 -20
- package/dist/task/spec-validation.js +17 -20
- package/dist/task/stall-detector.d.ts +23 -30
- package/dist/task/stall-detector.js +23 -30
- package/dist/task/stream-watchdog.d.ts +14 -12
- package/dist/task/stream-watchdog.js +14 -12
- package/dist/task/substitution-probe.d.ts +17 -20
- package/dist/task/substitution-probe.js +17 -20
- package/dist/task/task-gates.d.ts +36 -41
- package/dist/task/task-gates.js +95 -106
- package/dist/task/task-io.d.ts +4 -4
- package/dist/task/task-io.js +4 -4
- package/dist/task/task-parsers.js +4 -3
- package/dist/task/task-provenance.d.ts +2 -2
- package/dist/task/task-provenance.js +11 -13
- package/dist/task/task-types.d.ts +4 -3
- package/dist/task/terminal-outcome.d.ts +14 -16
- package/dist/task/terminal-outcome.js +12 -14
- package/dist/task/test-assembly.d.ts +13 -20
- package/dist/task/test-assembly.js +13 -20
- package/dist/task/timings.d.ts +5 -3
- package/dist/task/timings.js +5 -3
- package/dist/task/title-label.d.ts +9 -4
- package/dist/task/title-label.js +9 -4
- package/dist/task/type-only-answer.d.ts +44 -52
- package/dist/task/type-only-answer.js +44 -52
- package/dist/task/unfailable-command.d.ts +18 -24
- package/dist/task/unfailable-command.js +21 -27
- package/dist/task/unknown-routing.d.ts +10 -4
- package/dist/task/unknown-routing.js +10 -4
- package/dist/task/user-directives.d.ts +5 -8
- package/dist/task/user-directives.js +5 -8
- package/dist/task/verify-quality.d.ts +18 -22
- package/dist/task/verify-quality.js +45 -46
- package/dist/task/verify-reconcile.d.ts +15 -10
- package/dist/task/verify-reconcile.js +45 -43
- package/dist/task/verify-resolution.d.ts +24 -20
- package/dist/task/verify-resolution.js +51 -50
- package/dist/task/verify-work.d.ts +59 -66
- package/dist/task/verify-work.js +101 -138
- package/dist/task/widget.d.ts +15 -14
- package/dist/task/widget.js +22 -17
- package/dist/task/wiring-claims.d.ts +25 -32
- package/dist/task/wiring-claims.js +30 -35
- package/dist/task/write-guard.d.ts +39 -39
- package/dist/task/write-guard.js +48 -51
- package/dist/task/yolo.d.ts +34 -30
- package/dist/task/yolo.js +42 -37
- package/dist/workers/abstention.d.ts +21 -41
- package/dist/workers/abstention.js +27 -48
- package/dist/workers/brave-search.d.ts +4 -3
- package/dist/workers/brave-search.js +5 -2
- package/dist/workers/brave-warning.d.ts +7 -4
- package/dist/workers/brave-warning.js +19 -7
- package/dist/workers/ddg-search.d.ts +6 -6
- package/dist/workers/ddg-search.js +18 -12
- package/dist/workers/docs-cache.js +5 -2
- package/dist/workers/docs-chunk.d.ts +30 -37
- package/dist/workers/docs-chunk.js +37 -41
- package/dist/workers/docs-core.d.ts +28 -44
- package/dist/workers/docs-core.js +25 -44
- package/dist/workers/docs-index.js +4 -3
- package/dist/workers/docs-lookup.d.ts +15 -22
- package/dist/workers/docs-lookup.js +12 -21
- package/dist/workers/docs-project.d.ts +15 -9
- package/dist/workers/docs-project.js +17 -10
- package/dist/workers/docs-resolve.d.ts +19 -20
- package/dist/workers/docs-resolve.js +35 -32
- package/dist/workers/docs-retrieve.d.ts +5 -6
- package/dist/workers/docs-retrieve.js +18 -15
- package/dist/workers/exa-search.d.ts +9 -6
- package/dist/workers/exa-search.js +23 -12
- package/dist/workers/fetch-core.d.ts +13 -16
- package/dist/workers/fetch-core.js +23 -23
- package/dist/workers/focused-extractor.d.ts +12 -12
- package/dist/workers/focused-extractor.js +16 -19
- package/dist/workers/html-clean.js +24 -14
- package/dist/workers/http-request.d.ts +28 -20
- package/dist/workers/http-request.js +22 -17
- package/dist/workers/npm-version.d.ts +28 -11
- package/dist/workers/npm-version.js +24 -15
- package/dist/workers/phantom-imports.d.ts +15 -12
- package/dist/workers/phantom-imports.js +30 -24
- package/dist/workers/pi-worker-core.d.ts +86 -54
- package/dist/workers/pi-worker-core.js +112 -112
- package/dist/workers/pi-worker-docs.d.ts +24 -19
- package/dist/workers/pi-worker-docs.js +67 -76
- package/dist/workers/pi-worker-fetch.d.ts +7 -3
- package/dist/workers/pi-worker-fetch.js +27 -19
- package/dist/workers/pi-worker-search.js +12 -8
- package/dist/workers/pi-worker.d.ts +9 -4
- package/dist/workers/pi-worker.js +23 -10
- package/dist/workers/reasoning-warning.d.ts +18 -17
- package/dist/workers/reasoning-warning.js +22 -20
- package/dist/workers/research-cache.js +50 -78
- package/dist/workers/search-core.js +7 -5
- package/dist/workers/search-types.d.ts +10 -9
- package/dist/workers/search-types.js +9 -8
- package/dist/workers/session-hint.d.ts +13 -14
- package/dist/workers/session-hint.js +8 -9
- package/dist/workers/shared.d.ts +21 -25
- package/dist/workers/shared.js +0 -0
- package/dist/workers/single-read-extension.d.ts +14 -7
- package/dist/workers/single-read-extension.js +14 -7
- package/dist/workers/single-read-guard.d.ts +25 -28
- package/dist/workers/single-read-guard.js +32 -32
- package/dist/workers/typeonly-log.d.ts +12 -9
- package/dist/workers/typeonly-log.js +29 -33
- package/dist/workers/worker-channels.d.ts +15 -23
- package/dist/workers/worker-channels.js +15 -23
- package/dist/workers/worker-failure.d.ts +38 -46
- package/dist/workers/worker-failure.js +31 -39
- package/dist/workers/worker-kill.d.ts +25 -26
- package/dist/workers/worker-kill.js +16 -19
- package/dist/workers/worker-profiles.d.ts +43 -53
- package/dist/workers/worker-profiles.js +30 -38
- package/package.json +10 -8
|
@@ -1,31 +1,29 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The /task-plan interaction loop.
|
|
3
3
|
*
|
|
4
|
-
* Sequential
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
4
|
+
* Sequential and adaptive: ask ONE question at a time, feed the whole transcript
|
|
5
|
+
* back into the next generation call so later questions react to earlier answers,
|
|
6
|
+
* and stop when the model emits NONE. Generation, the question cap and the
|
|
7
|
+
* duplicate backstop live in question-source.ts; the answer cards and the A/B
|
|
8
|
+
* letter mapping in question-dialog.ts; markdown in inline-markdown.ts; the
|
|
9
|
+
* unattended policy in yolo.ts. /task's grill (phases.ts `phaseGrill`) and
|
|
10
|
+
* /task-auto's clarify (auto-orchestrator.ts `planAuto`) drive the same modules.
|
|
11
11
|
*
|
|
12
|
-
* What
|
|
13
|
-
*
|
|
14
|
-
*
|
|
12
|
+
* What this loop adds is the control surface. Grill and clarify let the user only
|
|
13
|
+
* answer the question in front of them. Here three moves are on every prompt, in
|
|
14
|
+
* this order of appearance:
|
|
15
15
|
*
|
|
16
16
|
* ❓ ask the model a question — the user asks, the model answers (PLAN_ASK)
|
|
17
|
-
* ✎ answer in your own words — the free-text card askQuestionBox
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
* doubles as "state a decision" when the model
|
|
21
|
-
* has nothing to ask
|
|
17
|
+
* ✎ answer in your own words — the free-text card askQuestionBox appends to
|
|
18
|
+
* every boxed picker. It doubles as "state a
|
|
19
|
+
* decision" when the model has nothing to ask.
|
|
22
20
|
* ▶ proceed to execution — stop planning, hand the decisions to /task
|
|
23
21
|
* (PLAN_PROCEED). Always the LAST card in the
|
|
24
22
|
* box — it ends the session, so it sits under
|
|
25
23
|
* every move that continues it, including the
|
|
26
24
|
* free-text card (see `manualPosition`).
|
|
27
25
|
*
|
|
28
|
-
* The loop
|
|
26
|
+
* The loop performs no I/O of its own: every side effect (child calls, dialogs,
|
|
29
27
|
* persistence) arrives through {@link PlanSessionDeps}, so the whole interaction
|
|
30
28
|
* is unit-testable without a TUI or a model.
|
|
31
29
|
*/
|
|
@@ -38,8 +36,8 @@ export { resolveAnswer } from './question-dialog.js';
|
|
|
38
36
|
// ─── Control actions ─────────────────────────────────────────────────────────
|
|
39
37
|
/**
|
|
40
38
|
* Sentinel values the picker resolves to when the user takes a control action
|
|
41
|
-
* instead of answering.
|
|
42
|
-
*
|
|
39
|
+
* instead of answering. Same shape as `USER_CANCELLED` (child-runner.ts): a
|
|
40
|
+
* value no model answer and no human ever types.
|
|
43
41
|
*/
|
|
44
42
|
export const PLAN_ASK = '__plan_ask__';
|
|
45
43
|
export const PLAN_PROCEED = '__plan_proceed__';
|
|
@@ -53,18 +51,19 @@ export const PLAN_STATE_LABEL = '✎ Add a decision of your own…';
|
|
|
53
51
|
export const PLAN_NO_QUESTIONS = 'No further questions — the decisions so far settle how this task is built.';
|
|
54
52
|
/**
|
|
55
53
|
* Hard ceiling on model-generated questions for one plan. The loop is open-ended
|
|
56
|
-
*
|
|
57
|
-
* does.
|
|
54
|
+
* — it stops when the model emits NONE — so this only bounds a model that never
|
|
55
|
+
* does. Same value as /task-auto's MAX_CLARIFY_QUESTIONS.
|
|
58
56
|
*/
|
|
59
57
|
export const MAX_PLAN_QUESTIONS = 8;
|
|
60
58
|
/**
|
|
61
59
|
* Corrective re-prompt for a question reply that did not follow the format —
|
|
62
|
-
* either nothing
|
|
63
|
-
* shape and same one-shot budget as GRILL_AUTO_FORMAT_HINT (prompts.ts)
|
|
64
|
-
*
|
|
65
|
-
*
|
|
66
|
-
*
|
|
67
|
-
*
|
|
60
|
+
* either nothing the parser could read, or a question with no `SUGGESTED:` line.
|
|
61
|
+
* Same shape and same one-shot budget as GRILL_AUTO_FORMAT_HINT (prompts.ts).
|
|
62
|
+
*
|
|
63
|
+
* Both failures are silent without it. `makeQuestionSource` answers `exhausted`
|
|
64
|
+
* for a reply it cannot parse, which this loop renders as "no further questions";
|
|
65
|
+
* and a question with no SUGGESTED reaches `buildOptionCards` with nothing to
|
|
66
|
+
* build a card from.
|
|
68
67
|
*/
|
|
69
68
|
export const PLAN_FORMAT_HINT = '[SYSTEM NOTE: Your previous reply did NOT follow the required format. Output exactly '
|
|
70
69
|
+ 'one numbered question line ("1. **...?** short rationale"), then on the NEXT line a '
|
|
@@ -72,16 +71,10 @@ export const PLAN_FORMAT_HINT = '[SYSTEM NOTE: Your previous reply did NOT follo
|
|
|
72
71
|
+ 'REQUIRED and must never be blank. Add an "ALT: " line only for a binary A-or-B fork. '
|
|
73
72
|
+ 'If nothing is left to ask, output the single token NONE and nothing else. No preamble, '
|
|
74
73
|
+ 'no analysis, no other text.]';
|
|
75
|
-
// Re-exported, NOT re-implemented. After the state machine moved to
|
|
76
|
-
// question-source.ts these were byte-identical copies: production read THAT
|
|
77
|
-
// module's, while plan-session.test.ts and four `scripts/live-*.ts` A/B harnesses
|
|
78
|
-
// read these — so a fix to `pickQuestion`'s heuristic would land in one copy while
|
|
79
|
-
// the harnesses kept measuring the other, and the measurement would silently stop
|
|
80
|
-
// describing shipped behaviour. That is the drift class this pass removes.
|
|
81
74
|
export { isNoneReply, pickQuestion } from './question-source.js';
|
|
82
75
|
/**
|
|
83
76
|
* Does the question offer the user a choice between two named alternatives?
|
|
84
|
-
* Deliberately shallow
|
|
77
|
+
* Deliberately shallow: an `or` anywhere before the first question mark.
|
|
85
78
|
*/
|
|
86
79
|
export function looksLikeFork(question) {
|
|
87
80
|
return /\bor\b/i.test(question.split('?')[0] ?? '');
|
|
@@ -89,21 +82,16 @@ export function looksLikeFork(question) {
|
|
|
89
82
|
/**
|
|
90
83
|
* Does the recommended default DEFER the decision instead of making one?
|
|
91
84
|
*
|
|
92
|
-
* The prompt asks for a "concrete, decisive default", and nothing enforced it.
|
|
93
|
-
* Live (aiz-client TASK_PLAN_0001, 2026-08-05): the model asked "what specific
|
|
94
|
-
* report should this new tab display?" and recommended
|
|
95
|
-
* "clarify with the user what the report is meant to show before proceeding".
|
|
96
|
-
* The user pressed enter, so it was recorded `(accepted recommendation)` and rode
|
|
97
|
-
* into /task's handoff as an AUTHORITATIVE decision — an order not to proceed,
|
|
98
|
-
* addressed to a run where no user exists. /task duly built a task whose
|
|
99
|
-
* ACCEPTANCE was "a planning document with placeholder sections" and whose VERIFY
|
|
100
|
-
* asserted that no source file had changed.
|
|
101
|
-
*
|
|
102
85
|
* A deferral is not an answer, and the one place it can never be one is here: the
|
|
103
86
|
* user IS present during planning, so "ask the user" is a null move — that IS the
|
|
104
|
-
* question.
|
|
105
|
-
*
|
|
106
|
-
*
|
|
87
|
+
* question. An accepted recommendation is a `decision` entry, so it rides into
|
|
88
|
+
* /task's handoff inside the block `buildHandoffPrompt` (plan-io.ts) labels
|
|
89
|
+
* authoritative — addressed to a run where no user exists.
|
|
90
|
+
*
|
|
91
|
+
* Every pattern is anchored to the START of the default, which keeps it off
|
|
92
|
+
* legitimate product behaviour: "prompt the user to confirm deletion" decides
|
|
93
|
+
* something and does not match; "ask the user which report" decides nothing and
|
|
94
|
+
* does.
|
|
107
95
|
*/
|
|
108
96
|
export function isDeferralSuggestion(suggested) {
|
|
109
97
|
const s = suggested.trim().replace(/^["'`*_\s]+/, '');
|
|
@@ -116,9 +104,9 @@ export function isDeferralSuggestion(suggested) {
|
|
|
116
104
|
|| /^(do not|don'?t|no)\b[^.]{0,40}\b(proceed|implement|build|start|write|decide)\b/i.test(s));
|
|
117
105
|
}
|
|
118
106
|
/**
|
|
119
|
-
* Corrective re-prompt for a default that deferred the decision.
|
|
120
|
-
*
|
|
121
|
-
*
|
|
107
|
+
* Corrective re-prompt for a default that deferred the decision. Quotes the
|
|
108
|
+
* question and the default back, because the child is a fresh process carrying
|
|
109
|
+
* only its prompt and cannot otherwise know what it just recommended.
|
|
122
110
|
*/
|
|
123
111
|
export function planDecisiveHint(question, suggested) {
|
|
124
112
|
return ('[SYSTEM NOTE: Your previous reply asked this question:\n'
|
|
@@ -133,17 +121,12 @@ export function planDecisiveHint(question, suggested) {
|
|
|
133
121
|
}
|
|
134
122
|
/**
|
|
135
123
|
* Corrective re-prompt for a fork-shaped question that shipped only ONE option.
|
|
124
|
+
* With no ALT, `buildOptionCards` emits a single card, so the user has to type
|
|
125
|
+
* out the alternative the model itself just named.
|
|
136
126
|
*
|
|
137
|
-
*
|
|
138
|
-
*
|
|
139
|
-
*
|
|
140
|
-
* type out the option the model itself had just proposed.
|
|
141
|
-
*
|
|
142
|
-
* The retry quotes the question back because the child is stateless (a fresh
|
|
143
|
-
* process per call, prompt only), so it cannot otherwise know what it just wrote.
|
|
144
|
-
* Validated before wiring (scripts/live-task-plan-fork-alt.ts): 6/6 fires
|
|
145
|
-
* recovered an ALT, and 6/6 re-asked the SAME question rather than changing the
|
|
146
|
-
* subject. It costs one extra child call on the questions where it fires.
|
|
127
|
+
* The retry quotes the question back because the child is a fresh process
|
|
128
|
+
* carrying only its prompt, so it cannot otherwise know what it just wrote. It
|
|
129
|
+
* costs one extra child call on the questions where it fires.
|
|
147
130
|
*/
|
|
148
131
|
export function planForkHint(question) {
|
|
149
132
|
return ('[SYSTEM NOTE: Your previous reply asked this question:\n'
|
|
@@ -158,11 +141,8 @@ export function planForkHint(question) {
|
|
|
158
141
|
* A default that DEFERS decides nothing, and an accepted deferral reaches the
|
|
159
142
|
* consumer dressed as an authoritative decision.
|
|
160
143
|
*
|
|
161
|
-
*
|
|
162
|
-
*
|
|
163
|
-
* literal with no compile link that this pass exists to remove — a rename would
|
|
164
|
-
* silently yield `[undefined]` and throw on the first clarify question of every
|
|
165
|
-
* run.
|
|
144
|
+
* Its own constant because it is the one rule BOTH tables below hold, shared by
|
|
145
|
+
* reference so the compiler links them.
|
|
166
146
|
*/
|
|
167
147
|
const DEFERRAL_RULE = {
|
|
168
148
|
id: 'SUGGESTED deferred the decision',
|
|
@@ -170,9 +150,10 @@ const DEFERRAL_RULE = {
|
|
|
170
150
|
planDecisiveHint(plain, q.suggested)
|
|
171
151
|
: null,
|
|
172
152
|
// When only the recommendation defers, the ALT is still a real commitment:
|
|
173
|
-
// promote it so the question keeps a usable default. With no ALT the
|
|
174
|
-
//
|
|
175
|
-
//
|
|
153
|
+
// promote it so the question keeps a usable default. With no ALT the default
|
|
154
|
+
// is dropped, `buildOptionCards` returns undefined, and `resolveAnswer` maps
|
|
155
|
+
// an empty submit to '(skipped)' rather than to a decision the user never
|
|
156
|
+
// made.
|
|
176
157
|
repair: q => {
|
|
177
158
|
const { alt: _alt, suggested: _suggested, ...rest } = q;
|
|
178
159
|
return q.alt !== undefined ? { ...rest, suggested: q.alt } : { ...rest };
|
|
@@ -181,16 +162,15 @@ const DEFERRAL_RULE = {
|
|
|
181
162
|
/**
|
|
182
163
|
* PLAN's quality rules, in order.
|
|
183
164
|
*
|
|
184
|
-
*
|
|
185
|
-
*
|
|
186
|
-
*
|
|
165
|
+
* The corrective-re-prompt budget is per QUESTION and shared across the whole
|
|
166
|
+
* table (question-source.ts): at most one rule fires per draw, and a defect that
|
|
167
|
+
* survives its re-prompt DEGRADES through `repair` rather than discarding the
|
|
168
|
+
* question — a weak default still beats no question. Each hint quotes the
|
|
169
|
+
* question back, because the child is a fresh process carrying only its prompt.
|
|
187
170
|
*
|
|
188
171
|
* Only {@link CLARIFY_QUALITY_RULES} is shared with `/task-auto`, and only the
|
|
189
|
-
* deferral rule is in it. The other two
|
|
190
|
-
*
|
|
191
|
-
* each costs one extra child call every time it fires, and clarify is the most
|
|
192
|
-
* A/B'd path in the codebase — moving them there is its own experiment, not a
|
|
193
|
-
* side effect of sharing a state machine. Recorded rather than done.
|
|
172
|
+
* deferral rule is in it. The other two cost an extra child call every time they
|
|
173
|
+
* fire.
|
|
194
174
|
*/
|
|
195
175
|
export const PLAN_QUALITY_RULES = [
|
|
196
176
|
{
|
|
@@ -202,7 +182,7 @@ export const PLAN_QUALITY_RULES = [
|
|
|
202
182
|
DEFERRAL_RULE,
|
|
203
183
|
{
|
|
204
184
|
// A fork-shaped question that ships one option leaves the user typing out
|
|
205
|
-
// the alternative the model itself just named
|
|
185
|
+
// the alternative the model itself just named.
|
|
206
186
|
id: 'fork-shaped question with no ALT',
|
|
207
187
|
detect: (q, plain) => q.alt === undefined && q.suggested !== undefined && looksLikeFork(plain) ?
|
|
208
188
|
planForkHint(plain)
|
|
@@ -212,12 +192,9 @@ export const PLAN_QUALITY_RULES = [
|
|
|
212
192
|
/**
|
|
213
193
|
* The deferral rule alone — the one clarify shares.
|
|
214
194
|
*
|
|
215
|
-
*
|
|
216
|
-
*
|
|
217
|
-
*
|
|
218
|
-
* asserted that no source file had changed. Clarify's answers ride into the
|
|
219
|
-
* decompose prompt and the AUTO file with exactly the same authority and had no
|
|
220
|
-
* guard at all — the same bug, one command over, waiting.
|
|
195
|
+
* Clarify's answers ride into the decompose prompt and the AUTO file with the
|
|
196
|
+
* same authority /task-plan's decisions ride into the handoff, so an accepted
|
|
197
|
+
* "clarify with the user before proceeding" lands there as an instruction too.
|
|
221
198
|
*
|
|
222
199
|
* It is also the only one of the three that costs nothing on the happy path: a
|
|
223
200
|
* decisive default never triggers it.
|
|
@@ -290,7 +267,8 @@ export const ASK_QUESTION = 'What do you want to ask about this task? The answer
|
|
|
290
267
|
export async function runPlanSession(deps) {
|
|
291
268
|
const entries = [];
|
|
292
269
|
const render = (s) => deps.renderMarkdown?.(s) ?? s;
|
|
293
|
-
/** The model has nothing (more) to ask: NONE, the
|
|
270
|
+
/** The model has nothing (more) to ask — any `exhausted` draw: NONE, the
|
|
271
|
+
* cap, the duplicate backstop, or a reply the parser could not read. */
|
|
294
272
|
let exhausted = false;
|
|
295
273
|
let pending = null;
|
|
296
274
|
const commit = async (entry) => {
|
|
@@ -1,37 +1,38 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* probe-gaming — deterministic detection of CHECK-GAMING code, feeding the verify
|
|
3
|
-
* and enforce
|
|
3
|
+
* gate's rule 4c (verify-work.ts) and the enforce prompt (enforce-guidelines.ts).
|
|
4
4
|
*
|
|
5
5
|
* The failure class: the implementation writes code — or a comment on it — whose
|
|
6
|
-
* STATED PURPOSE is to make a check pass, rather than to satisfy the requirement
|
|
7
|
-
* check stands for.
|
|
8
|
-
* `// Return 401 so the verification test passes
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
* its own intent it is cheaply detectable.
|
|
6
|
+
* STATED PURPOSE is to make a check pass, rather than to satisfy the requirement
|
|
7
|
+
* the check stands for. A catch-all handler carrying
|
|
8
|
+
* `// Return 401 so the verification test passes`
|
|
9
|
+
* leaves the real route dead while the check goes green. A check is a MESSENGER
|
|
10
|
+
* for a requirement; code written to quiet the messenger instead of meeting the
|
|
11
|
+
* requirement is the defect, and when it announces its own intent it is cheaply
|
|
12
|
+
* detectable.
|
|
14
13
|
*
|
|
15
|
-
* Same probe+rule design as the other
|
|
16
|
-
* prohibition-probe.ts, substitution-probe.ts, test-assembly.ts): a
|
|
17
|
-
*
|
|
18
|
-
* DIFF (added lines only — the work this
|
|
19
|
-
*
|
|
20
|
-
* rather than on the model
|
|
14
|
+
* Same probe+rule design as the other probes (skip-escape.ts,
|
|
15
|
+
* prohibition-probe.ts, substitution-probe.ts, test-assembly.ts): each is a
|
|
16
|
+
* `probeAdapter` row in verify-work.ts with its own findings, prompt block and
|
|
17
|
+
* numbered rule. This one scans the task's DIFF (added lines only — the work this
|
|
18
|
+
* task introduced) for the gaming tell and hands each hit to the gate child
|
|
19
|
+
* verbatim, so the rule fires on a concrete line rather than on the model
|
|
20
|
+
* self-discovering the intent.
|
|
21
21
|
*
|
|
22
|
-
* Advisory, never an auto-FAIL:
|
|
23
|
-
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
22
|
+
* Advisory, never an auto-FAIL: a probe contributes findings and a prompt block,
|
|
23
|
+
* and `probeAdapter` degrades an absent or throwing probe to its empty value.
|
|
24
|
+
* Intent phrasing is prose and a genuinely-benign line could in principle carry it
|
|
25
|
+
* ("returns 401 so the auth test passes" describing a CORRECT auth path). The child
|
|
26
|
+
* reads the exact line and the surrounding code and judges — the verify rule
|
|
27
|
+
* directs it to confirm the underlying requirement is actually met, not merely
|
|
28
|
+
* that the check is green.
|
|
27
29
|
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
34
|
-
* no such comment simply yields no findings (nothing to observe = pass).
|
|
30
|
+
* The tell is a purpose phrase binding a CHECK noun (test / verification / lint /
|
|
31
|
+
* CI / gate / …) to a pass-state or a gaming verb. Benign uses of the same words
|
|
32
|
+
* do not match: "pass the test data", "run the verification suite" and "the linter
|
|
33
|
+
* flagged this" all return false. Pure diff-text analysis — no stack, framework, or
|
|
34
|
+
* tool-name assumptions; a project with no such comment simply yields no findings,
|
|
35
|
+
* and nothing to observe is not a failure.
|
|
35
36
|
*/
|
|
36
37
|
/** One added line from a task's diff: the file it was added to and its text. */
|
|
37
38
|
export interface AddedLine {
|
|
Binary file
|
|
@@ -2,21 +2,18 @@
|
|
|
2
2
|
* prohibition-probe — deterministic detection of VIOLATED SPEC PROHIBITIONS,
|
|
3
3
|
* feeding the verify gate's prompt.
|
|
4
4
|
*
|
|
5
|
-
* The failure class
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
* diff at all and several affirmatively claimed the forbidden file was untouched.
|
|
5
|
+
* The failure class: the spec's CONSTRAINTS say "**Do NOT modify** any
|
|
6
|
+
* server-side code: `src/server/index.ts`, …", the implementation modifies that
|
|
7
|
+
* file anyway, and the verify child either waives the violation ("this is
|
|
8
|
+
* additive, tests pass with it") or never looks. Nothing else in the gate makes it
|
|
9
|
+
* run `git diff`, and rule 4b says it plainly: a forbidden file can be modified
|
|
10
|
+
* without any test noticing.
|
|
12
11
|
*
|
|
13
|
-
* So — like the substitution probe (
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
* collected for the substitution probe), and hand the child each hit with the
|
|
19
|
-
* exact constraint wording.
|
|
12
|
+
* So — like the substitution probe (substitution-probe.ts) — the fix is a
|
|
13
|
+
* pre-check whose finding is injected into the prompt: extract the concrete paths
|
|
14
|
+
* the spec forbids modifying, intersect with the task's changed files (pure git
|
|
15
|
+
* shape, from the same `collectChangedFiles` the substitution probe uses), and
|
|
16
|
+
* hand the child each hit with the exact constraint wording.
|
|
20
17
|
*
|
|
21
18
|
* The finding is advisory, not an auto-FAIL, for one reason: prohibitions in
|
|
22
19
|
* real specs are prose and can be CONDITIONAL ("Do NOT modify `api.ts` beyond
|
|
@@ -24,8 +21,9 @@
|
|
|
24
21
|
* false-FAIL legitimate work. The finding therefore carries the constraint line
|
|
25
22
|
* verbatim and the prompt's no-waiver rule (4b in verify-work.ts) forbids
|
|
26
23
|
* excusing an ABSOLUTE prohibition while directing conditional ones to be judged
|
|
27
|
-
* against their own stated exception. A violation
|
|
28
|
-
*
|
|
24
|
+
* against their own stated exception. A violation fully reverted before verify
|
|
25
|
+
* leaves no entry in `git diff HEAD`, so it never fires — reverted = not
|
|
26
|
+
* violated.
|
|
29
27
|
*/
|
|
30
28
|
import type { ChangedFile } from './substitution-probe.js';
|
|
31
29
|
/** One "do not modify X" constraint extracted from the spec text. */
|
package/dist/task/prompts.d.ts
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Prompt templates for every phase of the pi-task pipeline.
|
|
3
3
|
*
|
|
4
|
-
* Each template is a pure function
|
|
5
|
-
*
|
|
4
|
+
* Each template is a pure function or a plain string constant: inputs → prompt
|
|
5
|
+
* text. The module imports nothing, so there is no I/O and no side effect.
|
|
6
6
|
*/
|
|
7
7
|
export declare const MAX_GRILL_QUESTIONS = 20;
|
|
8
8
|
/**
|
|
@@ -21,8 +21,7 @@ export declare const COMPRESS_LABEL_PROMPT: (title: string, maxChars: number) =>
|
|
|
21
21
|
*
|
|
22
22
|
* Without it, refine is told "the task title is only a pointer into that spec —
|
|
23
23
|
* follow the spec" with no signal that the other steps exist, so a one-step
|
|
24
|
-
* "Scaffold …" title re-
|
|
25
|
-
* /task-auto run implemented all 24 steps under step 1).
|
|
24
|
+
* "Scaffold …" title can re-expand the entire design into one task.
|
|
26
25
|
*/
|
|
27
26
|
declare const REFINE_PROMPT: (raw: string, planContext?: string, existingFiles?: string, contracts?: string, directives?: string) => string;
|
|
28
27
|
declare const RESEARCH_READ_ONLY_CONSTRAINT = "IMPORTANT: You are ONLY allowed to READ. Do NOT create, modify, or delete any files. Use the read, grep, find, and ls tools to inspect the repo.";
|
package/dist/task/prompts.js
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Prompt templates for every phase of the pi-task pipeline.
|
|
3
3
|
*
|
|
4
|
-
* Each template is a pure function
|
|
5
|
-
*
|
|
4
|
+
* Each template is a pure function or a plain string constant: inputs → prompt
|
|
5
|
+
* text. The module imports nothing, so there is no I/O and no side effect.
|
|
6
6
|
*/
|
|
7
7
|
// Caps both the per-response question count (parseGrillQuestions/parseClarifyList)
|
|
8
8
|
// and the grill loop's total iterations, so a model that never emits NONE can't
|
|
@@ -33,8 +33,7 @@ ${title}`;
|
|
|
33
33
|
*
|
|
34
34
|
* Without it, refine is told "the task title is only a pointer into that spec —
|
|
35
35
|
* follow the spec" with no signal that the other steps exist, so a one-step
|
|
36
|
-
* "Scaffold …" title re-
|
|
37
|
-
* /task-auto run implemented all 24 steps under step 1).
|
|
36
|
+
* "Scaffold …" title can re-expand the entire design into one task.
|
|
38
37
|
*/
|
|
39
38
|
const REFINE_PROMPT = (raw, planContext, existingFiles, contracts, directives) => `${planContext ? planContext + '\n\n---\n\n' : ''}You receive a user's task description for an AI coding agent. Rewrite it to be unambiguous and actionable.
|
|
40
39
|
|
|
@@ -69,11 +68,11 @@ Task: ${raw}`;
|
|
|
69
68
|
const RESEARCH_READ_ONLY_CONSTRAINT = `IMPORTANT: You are ONLY allowed to READ. Do NOT create, modify, or delete any files. Use the read, grep, find, and ls tools to inspect the repo.`;
|
|
70
69
|
// Shared guard for every research worker. Open-ended tasks ("analyze the code",
|
|
71
70
|
// "how would you improve X", "write a report") tempt a worker into producing the
|
|
72
|
-
// deliverable itself —
|
|
73
|
-
// section
|
|
74
|
-
//
|
|
75
|
-
//
|
|
76
|
-
//
|
|
71
|
+
// deliverable itself — writing the whole code-review report into the CONTEXT
|
|
72
|
+
// section instead of the facts that section is for. Research only gathers INPUTS
|
|
73
|
+
// for a later spec; it must never be the deliverable. It also pins the output
|
|
74
|
+
// format: no preamble, no code fences, and no repeat of the section name as a
|
|
75
|
+
// header.
|
|
77
76
|
const RESEARCH_INPUTS_NOT_DELIVERABLE = `CRITICAL — you are gathering INPUTS for a later spec, NOT performing the task. Even if the task asks you to analyze, review, audit, report, plan, design, or write code, you must NOT produce that deliverable here. Do not write the report/analysis/plan/code. Your entire job is to emit the one structured section described below, which feeds a separate phase that writes the spec. Surveying the repo so that section is accurate is right; producing the task's output is wrong and wastes the run.
|
|
78
77
|
|
|
79
78
|
OUTPUT DISCIPLINE — strict: emit ONLY the raw section lines described below. No preamble (never "I've read the codebase…", never "Here is the … section"), no closing remarks, no Markdown headings, no code fences (no \`\`\`), and do NOT repeat the section name as a header. The first character of your output is the first entry of the list.`;
|
|
@@ -128,17 +127,8 @@ No section header. No other sections. No preamble.
|
|
|
128
127
|
|
|
129
128
|
Task:
|
|
130
129
|
${refined}`;
|
|
131
|
-
//
|
|
132
|
-
//
|
|
133
|
-
// package queries 20/20 vs 2/20, Fisher p ≈ 0 — confirming Stage 1's mechanism (completion is
|
|
134
|
-
// set by the output CONTRACT, not the answer or the target). But it failed invariant 2a: the
|
|
135
|
-
// ungrounded-symbol rate rose 1.0% -> 4.7% (p(rise) = 0.0010) and the semantics clause it adds
|
|
136
|
-
// carried 15% ungrounded symbols with the mandatory `UNVERIFIED:` abstention used 0 times in 40
|
|
137
|
-
// reps. The model obeys "ask a behaviour question" and ignores "abstain when you cannot verify",
|
|
138
|
-
// so it manufactures semantics — the exact F-1 laundering the file exists to prevent. The module
|
|
139
|
-
// and its tests are KEPT as the durable asset (like spec-urls.ts after PROMPT 4); only this
|
|
140
|
-
// interpolation is reverted. Full write-up: nexxtasks.txt "STAGE 2". Re-run: scripts/
|
|
141
|
-
// live-apis-contract-ab.ts. Do NOT re-wire without a lever that closes the abstention gap.
|
|
130
|
+
// `APIS_SEMANTICS_CONTRACT` (apis-contract.ts) is deliberately NOT interpolated
|
|
131
|
+
// into any prompt here. See that module's own docstring before wiring it in.
|
|
142
132
|
const RESEARCH_CONTEXT_PROMPT = (refined) => `You are doing targeted research for an AI coding agent. Use the read, grep, find, and ls tools to gather background knowledge and architectural context the agent will need for the following task.
|
|
143
133
|
|
|
144
134
|
RELEVANCE — read carefully: keep it tight. Each bullet must be an architectural fact that changes HOW the agent implements THIS task — a constraint, a non-obvious data flow, a gotcha, a hidden coupling. No general project tour, no restating the task, no facts the agent would not act on. If a bullet would not change a single implementation decision, drop it. There is no fixed bullet count — include every fact that bears on the task and no filler; fewer sharp bullets beat many shallow ones. If the task is itself an analysis or review, these bullets capture facts that analysis will rely on — they are NOT the analysis; do not write findings or recommendations here.
|
|
@@ -330,12 +320,13 @@ ${research}
|
|
|
330
320
|
User Q&A:
|
|
331
321
|
${qa}
|
|
332
322
|
${contracts && contracts.trim() ? `\n${contracts.trim()}\n` : ''}`;
|
|
333
|
-
// Fast triage pass run before the
|
|
334
|
-
//
|
|
335
|
-
//
|
|
336
|
-
//
|
|
337
|
-
//
|
|
338
|
-
//
|
|
323
|
+
// Fast triage pass run before the full rewrite. It produces either the single
|
|
324
|
+
// token CLEAN — meaning the compose draft needs no rewrite — or a short defect
|
|
325
|
+
// list. CLEAN alone does not skip the rewrite: phases.ts only short-circuits when
|
|
326
|
+
// the draft already parses a VERIFY block, no deterministic critique probe forced
|
|
327
|
+
// itself in, and there are no carried-in defects. Otherwise the triage defects are
|
|
328
|
+
// fed into CRITIQUE_PROMPT as a focus list so the rewrite targets real problems
|
|
329
|
+
// instead of re-deriving them from scratch.
|
|
339
330
|
const CRITIQUE_TRIAGE_PROMPT = (spec, refined, qa, contracts) => `You are triaging an implementation spec for an AI coding agent. Decide whether it needs a rewrite. Do NOT rewrite it — only judge it.
|
|
340
331
|
|
|
341
332
|
The refined task and the user's Q&A below are GROUND TRUTH. Judge the spec against them. Look for SUBSTANTIVE defects only:
|
|
@@ -3,25 +3,18 @@
|
|
|
3
3
|
* (`/task-auto`) — and the one place its numbering, its formatting and its
|
|
4
4
|
* provenance policy live.
|
|
5
5
|
*
|
|
6
|
-
* `question-dialog.ts`
|
|
7
|
-
*
|
|
8
|
-
* eight retyped push sites across two files, each choosing a suffix by hand
|
|
9
|
-
* according to which branch of an `if/else` it stood in.
|
|
6
|
+
* `question-dialog.ts` owns the ANSWER side of these loops — the picker cards and
|
|
7
|
+
* the reply mapping. This owns what the loop RECORDS.
|
|
10
8
|
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
* is
|
|
14
|
-
*
|
|
15
|
-
* the YOLO branch pushes `${answer} ${YOLO_STAMP}` into that very array. The rule
|
|
16
|
-
* was violated inside the loop body that declares it, because the rule lived in
|
|
17
|
-
* prose and the decision lived at each push.
|
|
18
|
-
*
|
|
19
|
-
* Here it is a property of a value: an entry states its KIND, and the policy says
|
|
20
|
-
* where that kind's provenance is allowed to appear.
|
|
9
|
+
* Provenance is a property of a value here, not a decision taken at each push
|
|
10
|
+
* site: an entry states its KIND, and the policy says where that kind's
|
|
11
|
+
* provenance is allowed to appear. The two renderings below differ ONLY in those
|
|
12
|
+
* suffixes, which is how a hand-stamped transcript drifts apart unnoticed.
|
|
21
13
|
*
|
|
22
14
|
* NOT unified here: `plan-session.ts`'s `PlanEntry` transcript. It is a different
|
|
23
|
-
* shape — decisions
|
|
24
|
-
*
|
|
15
|
+
* shape — model decisions, advisory notes and user-stated decisions in one list,
|
|
16
|
+
* persisted to its own task file — and folding it in would be a rename, not a
|
|
17
|
+
* deepening.
|
|
25
18
|
*/
|
|
26
19
|
/** How one answer was arrived at. */
|
|
27
20
|
export type QaKind =
|
|
@@ -56,7 +49,6 @@ export interface QaPolicy {
|
|
|
56
49
|
* becomes model input describing how the answer was obtained rather than what
|
|
57
50
|
* it was. Clarify deliberately shows its generator the provenance, so a
|
|
58
51
|
* question the triage already settled reads as settled and is not re-asked.
|
|
59
|
-
* An option, not a unification — it is observable either way.
|
|
60
52
|
*/
|
|
61
53
|
generatorSeesProvenance: boolean;
|
|
62
54
|
}
|
|
@@ -64,10 +56,11 @@ export interface QaPolicy {
|
|
|
64
56
|
* GRILL: stamps `auto` and both YOLO kinds in the record, and shows the generator
|
|
65
57
|
* nothing.
|
|
66
58
|
*
|
|
67
|
-
* `accepted` is deliberately NOT in the record set
|
|
68
|
-
* `COMPOSE_PROMPT` and `CRITIQUE_PROMPT
|
|
69
|
-
*
|
|
70
|
-
*
|
|
59
|
+
* `accepted` is deliberately NOT in the record set, and clarify's policy below
|
|
60
|
+
* disagrees. Grill's record is handed to `COMPOSE_PROMPT` and `CRITIQUE_PROMPT`
|
|
61
|
+
* (the latter names the Q&A GROUND TRUTH); clarify's is handed to
|
|
62
|
+
* `AUTO_DECOMPOSE_PROMPT`. Whether the two should agree is a question about those
|
|
63
|
+
* prompts, so each keeps its own answer here.
|
|
71
64
|
*/
|
|
72
65
|
export declare const GRILL_QA_POLICY: QaPolicy;
|
|
73
66
|
/** CLARIFY: stamps every non-typed kind, in the record and to the generator alike. */
|
|
@@ -90,7 +83,7 @@ export declare class QaTranscript {
|
|
|
90
83
|
/** How many questions have been answered. Also the next question's number − 1. */
|
|
91
84
|
get length(): number;
|
|
92
85
|
get entries(): ReadonlyArray<QaEntry>;
|
|
93
|
-
/** Record one answered question.
|
|
86
|
+
/** Record one answered question. Position IS the number; no caller supplies one. */
|
|
94
87
|
add(kind: QaKind, question: string, answer: string): void;
|
|
95
88
|
/** The persisted / handed-on transcript, with provenance per the policy. */
|
|
96
89
|
forRecord(): string;
|