pi-ui-extend 1.0.39 → 1.0.41

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. package/README.md +1 -1
  2. package/dist/app/commands/command-registry.js +2 -2
  3. package/dist/app/commands/command-session-actions.d.ts +0 -1
  4. package/dist/app/commands/command-session-actions.js +22 -13
  5. package/dist/app/icons.d.ts +14 -0
  6. package/dist/app/icons.js +33 -0
  7. package/dist/app/rendering/conversation-tool-renderer.js +2 -2
  8. package/dist/app/rendering/dcp-stats.d.ts +6 -1
  9. package/dist/app/rendering/dcp-stats.js +214 -46
  10. package/dist/app/rendering/editor-panels.js +8 -5
  11. package/dist/app/session/lazy-session-manager.js +12 -1
  12. package/dist/app/session/tabs-controller.d.ts +2 -5
  13. package/dist/app/session/tabs-controller.js +12 -21
  14. package/dist/app/subagents/subagents-model.d.ts +14 -1
  15. package/dist/app/subagents/subagents-model.js +34 -15
  16. package/dist/app/types.d.ts +2 -0
  17. package/dist/bundled-extensions/session-title/config.js +1 -1
  18. package/dist/markdown-format.js +27 -9
  19. package/dist/schemas/pi-tools-suite-schema.d.ts +29 -16
  20. package/dist/schemas/pi-tools-suite-schema.js +46 -31
  21. package/external/pi-tools-suite/README.md +392 -52
  22. package/external/pi-tools-suite/docs/browser-qa-subagent.md +31 -21
  23. package/external/pi-tools-suite/docs/context-gateway-p00-adr.md +216 -0
  24. package/external/pi-tools-suite/docs/context-gateway-p01n-gate-review.md +122 -0
  25. package/external/pi-tools-suite/docs/context-gateway-p01n-measurement.md +133 -0
  26. package/external/pi-tools-suite/docs/context-gateway-p01r-ra-evidence.md +111 -0
  27. package/external/pi-tools-suite/docs/context-gateway-p01r-rb-evidence.md +100 -0
  28. package/external/pi-tools-suite/docs/context-gateway-p01r-rc-evidence.md +69 -0
  29. package/external/pi-tools-suite/docs/context-gateway-p01r-rd-evidence.md +100 -0
  30. package/external/pi-tools-suite/docs/context-gateway-p01r-re-evidence.md +74 -0
  31. package/external/pi-tools-suite/docs/context-gateway-p01r-rf-evidence.md +153 -0
  32. package/external/pi-tools-suite/docs/context-gateway-p01r-rg-evidence.md +235 -0
  33. package/external/pi-tools-suite/docs/evals.md +684 -0
  34. package/external/pi-tools-suite/docs/subagent-model-pools.md +109 -0
  35. package/external/pi-tools-suite/package.json +10 -3
  36. package/external/pi-tools-suite/src/async-subagents/{private-skills → agents}/browser-qa/scripts/browser-qa-runner.mjs +82 -1
  37. package/external/pi-tools-suite/src/async-subagents/{private-skills/browser-qa/SKILL.md → agents/browser-qa.md} +261 -12
  38. package/external/pi-tools-suite/src/async-subagents/agents/implement.md +20 -0
  39. package/external/pi-tools-suite/src/async-subagents/agents/oracle.md +16 -0
  40. package/external/pi-tools-suite/src/async-subagents/agents/research.md +18 -0
  41. package/external/pi-tools-suite/src/async-subagents/agents/verify.md +18 -0
  42. package/external/pi-tools-suite/src/async-subagents/async-subagents.sample.jsonc +27 -243
  43. package/external/pi-tools-suite/src/async-subagents/commands.ts +6 -2
  44. package/external/pi-tools-suite/src/async-subagents/core/agent-catalog.ts +41 -0
  45. package/external/pi-tools-suite/src/async-subagents/core/agent-strategy.ts +13 -93
  46. package/external/pi-tools-suite/src/async-subagents/core/agents-dir.ts +494 -0
  47. package/external/pi-tools-suite/src/async-subagents/core/browser-qa.ts +9 -0
  48. package/external/pi-tools-suite/src/async-subagents/core/config.ts +200 -143
  49. package/external/pi-tools-suite/src/async-subagents/core/model-fallback.ts +1 -1
  50. package/external/pi-tools-suite/src/async-subagents/core/model-selection.ts +54 -0
  51. package/external/pi-tools-suite/src/async-subagents/core/prompt.ts +7 -6
  52. package/external/pi-tools-suite/src/async-subagents/core/routing.ts +52 -45
  53. package/external/pi-tools-suite/src/async-subagents/core/spawn.ts +12 -4
  54. package/external/pi-tools-suite/src/async-subagents/index.ts +11 -1
  55. package/external/pi-tools-suite/src/async-subagents/lib.ts +6 -2
  56. package/external/pi-tools-suite/src/async-subagents/tools/spawn.ts +46 -18
  57. package/external/pi-tools-suite/src/async-subagents/tools/subagents.ts +3 -2
  58. package/external/pi-tools-suite/src/async-subagents/types.ts +2 -0
  59. package/external/pi-tools-suite/src/coding-discipline/index.ts +41 -142
  60. package/external/pi-tools-suite/src/config.ts +1 -22
  61. package/external/pi-tools-suite/src/context-gateway/accounting.ts +151 -0
  62. package/external/pi-tools-suite/src/context-gateway/config.ts +111 -0
  63. package/external/pi-tools-suite/src/context-gateway/index.ts +160 -0
  64. package/external/pi-tools-suite/src/context-gateway/metadata-normalization.ts +88 -0
  65. package/external/pi-tools-suite/src/context-gateway/storeless-capabilities.ts +89 -0
  66. package/external/pi-tools-suite/src/context-gateway/telemetry.ts +429 -0
  67. package/external/pi-tools-suite/src/context-gateway/test-output-parser.ts +326 -0
  68. package/external/pi-tools-suite/src/context-gateway/types.ts +152 -0
  69. package/external/pi-tools-suite/src/dcp/auto-compress-budget.ts +106 -0
  70. package/external/pi-tools-suite/src/dcp/auto-compress.ts +810 -106
  71. package/external/pi-tools-suite/src/dcp/commands.ts +64 -139
  72. package/external/pi-tools-suite/src/dcp/compress-tool.ts +369 -35
  73. package/external/pi-tools-suite/src/dcp/compression-blocks.ts +510 -64
  74. package/external/pi-tools-suite/src/dcp/compression-preview.ts +113 -0
  75. package/external/pi-tools-suite/src/dcp/compression-progress.ts +70 -0
  76. package/external/pi-tools-suite/src/dcp/config.ts +36 -61
  77. package/external/pi-tools-suite/src/dcp/conversation-index.ts +421 -0
  78. package/external/pi-tools-suite/src/dcp/debug-log.ts +7 -5
  79. package/external/pi-tools-suite/src/dcp/index.ts +617 -203
  80. package/external/pi-tools-suite/src/dcp/journal.ts +566 -0
  81. package/external/pi-tools-suite/src/dcp/progress-controller.ts +244 -0
  82. package/external/pi-tools-suite/src/dcp/prompts.ts +10 -7
  83. package/external/pi-tools-suite/src/dcp/provider-tool-results.ts +189 -0
  84. package/external/pi-tools-suite/src/dcp/pruner-candidates.ts +298 -78
  85. package/external/pi-tools-suite/src/dcp/pruner-compression-blocks.ts +173 -281
  86. package/external/pi-tools-suite/src/dcp/pruner-emergency.ts +2 -4
  87. package/external/pi-tools-suite/src/dcp/pruner-message-ids.ts +17 -5
  88. package/external/pi-tools-suite/src/dcp/pruner-metadata.ts +11 -1
  89. package/external/pi-tools-suite/src/dcp/pruner-nudge.ts +30 -82
  90. package/external/pi-tools-suite/src/dcp/pruner-tools.ts +22 -133
  91. package/external/pi-tools-suite/src/dcp/pruner.ts +18 -33
  92. package/external/pi-tools-suite/src/dcp/recovery.ts +129 -0
  93. package/external/pi-tools-suite/src/dcp/shadow-plan.ts +127 -0
  94. package/external/pi-tools-suite/src/dcp/state-transaction.ts +102 -0
  95. package/external/pi-tools-suite/src/dcp/state.ts +158 -580
  96. package/external/pi-tools-suite/src/dcp/ui.ts +1 -0
  97. package/external/pi-tools-suite/src/default-pi-tools-suite-config.ts +55 -220
  98. package/external/pi-tools-suite/src/index.ts +9 -0
  99. package/external/pi-tools-suite/src/model-tools/index.ts +76 -42
  100. package/external/pi-tools-suite/src/repo-discovery/index.ts +84 -18
  101. package/external/pi-tools-suite/src/repo-discovery/native-compact.ts +458 -0
  102. package/external/pi-tools-suite/src/session-recovery/index.ts +189 -43
  103. package/external/pi-tools-suite/src/tool-descriptions.ts +43 -38
  104. package/external/pi-tools-suite/src/truncation-metadata-normalizer/index.ts +17 -0
  105. package/package.json +6 -6
  106. package/schemas/pi-tools-suite.json +159 -78
  107. package/external/pi-tools-suite/src/async-subagents/private-skills/browser-qa/references/auth-scaffold-spec.md +0 -78
  108. package/external/pi-tools-suite/src/async-subagents/private-skills/browser-qa/references/qa-design.md +0 -223
  109. package/external/pi-tools-suite/src/dcp/state-persistence.ts +0 -195
  110. /package/external/pi-tools-suite/src/async-subagents/{private-skills/browser-qa/references → agents/browser-qa/examples}/qa-auth.example.jsonc +0 -0
  111. /package/external/pi-tools-suite/src/async-subagents/{private-skills/browser-qa/references → agents/browser-qa/examples}/qa-flow.example.jsonc +0 -0
  112. /package/external/pi-tools-suite/src/async-subagents/{private-skills → agents}/browser-qa/vendor/fflate.LICENSE +0 -0
  113. /package/external/pi-tools-suite/src/async-subagents/{private-skills → agents}/browser-qa/vendor/fflate.mjs +0 -0
@@ -0,0 +1,109 @@
1
+ # Sub-agent model pools
2
+
3
+ Sub-agents primarily reduce the cost of bounded work and keep intermediate
4
+ source, searches and logs out of the parent context. The parent owns planning,
5
+ integration, decisions and the final answer. Actual savings depend on worker
6
+ quality, retries and how much work the parent repeats; the configuration is
7
+ not a price oracle.
8
+
9
+ ## Five execution modes
10
+
11
+ - `research`: read-only evidence gathering, searches and independent diff review.
12
+ - `implement`: bounded code, documentation, test and frontend changes.
13
+ - `verify`: run checks and interpret logs, without fixing source or tests.
14
+ - `browser-qa`: isolated browser workflow with assertions and visual artifacts.
15
+ - `oracle`: a deliberate strong second opinion, not automatic worker escalation.
16
+
17
+ Task-specific discipline belongs in the brief or `promptAppend`. A new project
18
+ agent is warranted when it adds a durable contract, capabilities or resources,
19
+ not merely a professional title. `verify` has a behavioral no-edit contract;
20
+ shell access is not a read-only filesystem sandbox.
21
+
22
+ ## Agent priority, preset availability
23
+
24
+ Each Markdown profile owns its ordered `models` list. This is one candidate
25
+ chain for initial selection and subsequent quota fallbacks:
26
+
27
+ ```yaml
28
+ ---
29
+ description: Make bounded implementation changes.
30
+ models:
31
+ - zai/glm-5.3-flash
32
+ - openai-codex/gpt-5.6-terra
33
+ - openai-codex/gpt-5.6-luna
34
+ thinking: medium
35
+ ---
36
+ ```
37
+
38
+ A preset contains a set of available models, not a per-agent matrix:
39
+
40
+ ```jsonc
41
+ {
42
+ "asyncSubagents": {
43
+ "presets": {
44
+ "gpt": {
45
+ "description": "Models available for this session",
46
+ "models": [
47
+ "openai-codex/gpt-5.6-luna",
48
+ "openai-codex/gpt-5.6-terra",
49
+ "openai-codex/gpt-5.6-sol"
50
+ ]
51
+ }
52
+ }
53
+ }
54
+ }
55
+ ```
56
+
57
+ The example agent selects Terra, not Luna: the agent's order wins. Sol is
58
+ available in the pool but absent from this worker's chain, so it cannot become
59
+ an automatic implementation fallback. The oracle can declare Sol in its own
60
+ chain. Model references must be exact `provider/model` values, not wildcards.
61
+
62
+ The resolver intersects the agent chain with the selected pool. Runtime
63
+ selection then skips unregistered, unauthenticated or session-exhausted models.
64
+ Tasks with images and browser QA require confirmed image support. The first
65
+ eligible candidate runs; only the remaining eligible candidates are passed to
66
+ quota fallback. An empty intersection or unavailable chain rejects the batch
67
+ before any children or run state are created. Model selection makes no LLM
68
+ completion request; the optional role router is a separate operation.
69
+
70
+ Oracle prefers another provider when possible, but still respects the pool.
71
+ It never substitutes an ordinary cheap candidate merely to avoid a selection
72
+ error. A single-provider pool cannot promise cross-provider independence.
73
+
74
+ Explicit task `model`, CLI `--model`, and `FORCE_CURRENT_MODEL` remain deliberate
75
+ overrides: they bypass the pool and do not add automatic fallback candidates.
76
+ The parent should not use these to evade the configured budget. The pool is
77
+ a selection policy, not a security boundary against explicit overrides.
78
+
79
+ ## Selection and compatibility
80
+
81
+ Use `/subagent-preset <name>`, `AGENTS_PRESET=<name>` or
82
+ `/subagent-preset session <name>`. Clearing the preset uses agent priorities
83
+ without a pool filter. The shipped names remain compatible with saved choices:
84
+ `cheap` is the GLM pool, `gpt` the GPT pool, and `deep` the mixed pool. The last
85
+ name no longer means that ordinary workers should escalate to flagship models.
86
+
87
+ Old role names are not implicit aliases. `quick`, `scan`, `review`, `deep`,
88
+ `docs`, `frontend`, and `tests` work only when explicitly defined as ordinary
89
+ custom/project types. This keeps the effective catalog and accepted names exact.
90
+
91
+ Legacy `model` plus `fallbackModels` and `modelByParent` still load. New profile
92
+ `models` replaces inherited legacy selection fields; an explicit legacy model
93
+ override can still replace an inherited new list. Empty `models` means no
94
+ candidates, not permission to inherit the parent model. Model-less project
95
+ specialists must declare candidates or receive an explicit model override.
96
+
97
+ Legacy preset role matrices remain readable. A preset with `models` uses only
98
+ the pool contract, dropping stale legacy model/thinking/type overrides. Switching
99
+ a higher-priority config layer back to a legacy preset removes the inherited
100
+ pool. Configuration loading never rewrites user files; review old overrides
101
+ when migrating, since explicitly saved profiles can retain expensive models.
102
+
103
+ ## Compact handoff
104
+
105
+ Give workers a scope, acceptance criteria and the evidence needed to start.
106
+ Read compact results first and inspect raw artifacts selectively. One noisy
107
+ sequential investigation can justify a worker; a command whose exit status is
108
+ sufficient usually only needs a saved log, not another LLM. Independent review
109
+ uses a fresh `research` invocation, not a separate built-in persona.
@@ -29,6 +29,13 @@
29
29
  "test:prompt-evals:async": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/async-subagents/selection-e2e.test.ts test/prompt-evals/async-routing-e2e.test.ts",
30
30
  "test:prompt-evals:dcp": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/prompt-evals/dcp-summary-e2e.test.ts",
31
31
  "test:prompt-evals": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/tool-selection-e2e.test.ts test/async-subagents/selection-e2e.test.ts test/prompt-evals",
32
+ "test:evals:contracts": "bun test test/evals/extension-contracts.test.ts test/evals/harness.test.ts",
33
+ "test:evals:live": "PI_TOOLS_SUITE_EVALS_LIVE=1 bun test --concurrent --max-concurrency=4 test/evals/live-evals.test.ts",
34
+ "evals": "npm run test:evals:contracts && bun test test/evals/live-evals.test.ts",
35
+ "evals:report": "bun test/evals/run-evals.ts",
36
+ "evals:p01n": "bun test/evals/run-p01n-paired.ts",
37
+ "evals:context-gateway-observe": "bun test/evals/run-context-gateway-observe.ts",
38
+ "evals:context-gateway-recovery": "bun test/evals/run-context-gateway-recovery.ts",
32
39
  "bench:locate": "PI_LOCATE_BENCH_ITERATIONS=5 PI_LOCATE_BENCH_FAKE_IDX=0 PI_LOCATE_BENCH_MODEL=zai/glm-5-turbo PI_LOCATE_BENCH_MODES=direct-read-grep,ast-structural,repo-search-hybrid,repo-discovery,subagent-search,unrestricted-suite node test/fixtures/hard-to-find-project/benchmark/run-locate-benchmark.mjs",
33
40
  "bench:locate:analyze": "node test/fixtures/hard-to-find-project/benchmark/analyze-locate-benchmark.mjs",
34
41
  "test:locate-benchmark-e2e": "PI_LOCATE_BENCH_E2E=1 PI_LOCATE_BENCH_MODEL=zai/glm-5-turbo bun test test/locate-benchmark-e2e.test.ts",
@@ -44,9 +51,9 @@
44
51
  "vscode-languageserver-protocol": "^3.17.5"
45
52
  },
46
53
  "peerDependencies": {
47
- "@earendil-works/pi-ai": "0.85.0",
48
- "@earendil-works/pi-coding-agent": "0.85.0",
49
- "@earendil-works/pi-tui": "0.85.0",
54
+ "@earendil-works/pi-ai": "0.85.1",
55
+ "@earendil-works/pi-coding-agent": "0.85.1",
56
+ "@earendil-works/pi-tui": "0.85.1",
50
57
  "typebox": "*"
51
58
  },
52
59
  "devDependencies": {
@@ -2206,12 +2206,60 @@ foreach ($candidatePid in $descendants) {
2206
2206
  });
2207
2207
  return;
2208
2208
  }
2209
+
2210
+ // Prefer parent-based enumeration when available. macOS sandboxed hosts can
2211
+ // deny `ps` while still allowing pgrep(1); relying only on `ps` leaves a
2212
+ // detached Playwright/browser process alive after a timeout. Collect the
2213
+ // whole tree before killing anything so descendants cannot be re-parented
2214
+ // out of reach, then terminate deepest-first. A detached child is normally a
2215
+ // process-group leader, so also try its negative PID before the individual
2216
+ // PID; non-group-leaders simply yield ESRCH and fall through safely.
2217
+ const pgrepDescendants = [];
2218
+ const pgrepPendingParents = [process.pid];
2219
+ let pgrepAvailable = true;
2220
+ for (let index = 0; index < pgrepPendingParents.length; index += 1) {
2221
+ const parentPid = pgrepPendingParents[index];
2222
+ const result = spawnSync("pgrep", ["-P", String(parentPid)], {
2223
+ encoding: "utf8",
2224
+ timeout: 1000,
2225
+ maxBuffer: 256 * 1024,
2226
+ });
2227
+ if (result.error?.code === "ENOENT") {
2228
+ pgrepAvailable = false;
2229
+ break;
2230
+ }
2231
+ if (result.status !== 0 && result.status !== 1) {
2232
+ pgrepAvailable = false;
2233
+ break;
2234
+ }
2235
+ for (const line of (result.stdout ?? "").split("\n")) {
2236
+ const pid = Number(line.trim());
2237
+ if (!Number.isInteger(pid) || pid <= 0 || pgrepDescendants.includes(pid)) continue;
2238
+ pgrepDescendants.push(pid);
2239
+ pgrepPendingParents.push(pid);
2240
+ }
2241
+ }
2242
+ if (pgrepAvailable) {
2243
+ for (const pid of [...pgrepDescendants].reverse()) {
2244
+ try { process.kill(-pid, "SIGKILL"); } catch { /* not a group leader or already gone */ }
2245
+ try { process.kill(pid, "SIGKILL"); } catch { /* already gone */ }
2246
+ }
2247
+ return;
2248
+ }
2249
+
2209
2250
  const snapshot = spawnSync("ps", ["-axo", "pid=,ppid=,pgid="], {
2210
2251
  encoding: "utf8",
2211
2252
  timeout: 1000,
2212
2253
  maxBuffer: 2 * 1024 * 1024,
2213
2254
  });
2214
- if (snapshot.status !== 0 || !snapshot.stdout) return;
2255
+ if (snapshot.status !== 0 || !snapshot.stdout) {
2256
+ // Sandboxed macOS hosts can deny `ps` process enumeration while still
2257
+ // allowing parent-scoped `pgrep -P`. Fall back to a bounded recursive
2258
+ // parent walk so a detached browser process is not left behind merely
2259
+ // because the broader process snapshot is unavailable.
2260
+ terminateRunnerDescendantsWithPgrep(process.pid);
2261
+ return;
2262
+ }
2215
2263
  const processes = snapshot.stdout
2216
2264
  .split("\n")
2217
2265
  .map((line) => line.trim().split(/\s+/).map(Number))
@@ -2241,6 +2289,39 @@ foreach ($candidatePid in $descendants) {
2241
2289
  }
2242
2290
  }
2243
2291
 
2292
+ function terminateRunnerDescendantsWithPgrep(rootPid) {
2293
+ const descendants = [];
2294
+ const seen = new Set([rootPid]);
2295
+ const pendingParents = [rootPid];
2296
+ for (let index = 0; index < pendingParents.length && descendants.length < 4096; index += 1) {
2297
+ const parentPid = pendingParents[index];
2298
+ const result = spawnSync("pgrep", ["-P", String(parentPid)], {
2299
+ encoding: "utf8",
2300
+ timeout: 500,
2301
+ maxBuffer: 256 * 1024,
2302
+ });
2303
+ // pgrep exits 1 when there are no matching children. Any other failure is
2304
+ // fail-closed for this fallback; callers still terminate the runner itself.
2305
+ if (result.status !== 0 && result.status !== 1) continue;
2306
+ for (const line of (result.stdout ?? "").split("\n")) {
2307
+ const pid = Number.parseInt(line.trim(), 10);
2308
+ if (!Number.isInteger(pid) || pid <= 0 || seen.has(pid)) continue;
2309
+ seen.add(pid);
2310
+ descendants.push(pid);
2311
+ pendingParents.push(pid);
2312
+ }
2313
+ }
2314
+
2315
+ // Kill leaves before their parents so descendants cannot be reparented
2316
+ // before we have sent them a signal. Detached children are killed by PID;
2317
+ // unlike the primary `ps` path this fallback does not need their PGID.
2318
+ for (let index = descendants.length - 1; index >= 0; index -= 1) {
2319
+ try {
2320
+ process.kill(descendants[index], "SIGKILL");
2321
+ } catch { /* already gone */ }
2322
+ }
2323
+ }
2324
+
2244
2325
  function sanitizeTraceArchiveInWorker(tracePath, secrets, deadline) {
2245
2326
  return new Promise((resolve, reject) => {
2246
2327
  const worker = new Worker(new URL(import.meta.url), {
@@ -1,15 +1,25 @@
1
1
  ---
2
- name: browser-qa-private
3
- description: Private self-contained workflow for deterministic browser bug reproduction and fix verification with redacted project auth and Playwright evidence.
2
+ description: Use for browser-based visual QA - reproduce UI bugs and verify fixes with deterministic assertions, screenshots, video, and traces.
3
+ icon: globe
4
+ models: [zai/glm-5.3-flash, openai-codex/gpt-5.6-luna]
5
+ thinking: low
6
+ timeoutMs: 300000
7
+ tools: [read, grep, bash]
4
8
  ---
5
9
 
6
10
  # Browser QA
7
11
 
8
- Use this skill's bundled runner as the only browser interface. It already owns
12
+ Use this agent's bundled runner as the only browser interface. It already owns
9
13
  Playwright, browser/context lifecycle, tracing, video, screenshots, origin
10
14
  isolation, authentication, redaction, and cleanup. Do not invoke another browser
11
15
  CLI, create shared/default browser sessions, or generate executable browser code.
12
16
 
17
+ The launcher sets `PI_BROWSER_QA_RUNNER` to the absolute path of the installed
18
+ runner. Invoke it as `node "$PI_BROWSER_QA_RUNNER"`; do not guess a path from
19
+ the project cwd, override this variable, or copy/reimplement the runner.
20
+ The complete workflow and scenario-design guidance are included below. No
21
+ additional skill or instruction-file discovery is needed.
22
+
13
23
  Never read, print, grep, copy, or edit credential values from
14
24
  `.pi/qa_auth.jsonc` yourself.
15
25
 
@@ -37,7 +47,8 @@ inside the preflight, return a structured `BLOCKED` result immediately. Do not
37
47
  consume the launcher budget on open-ended source reading, server polling,
38
48
  capability probing, or retries.
39
49
 
40
- 1. Resolve `scripts/browser-qa-runner.mjs` relative to this skill.
50
+ 1. Use the launcher-provided `PI_BROWSER_QA_RUNNER` path. If it is missing,
51
+ report a launcher configuration blocker rather than selecting another runner.
41
52
  2. Use the launcher-provided `$PI_SUBAGENT_AGENT_DIR/browser-qa/` workspace.
42
53
  The launcher creates its private `flows/` directory and the runner rejects
43
54
  flows or evidence destinations outside this owning sub-agent directory. Do
@@ -48,7 +59,7 @@ capability probing, or retries.
48
59
  be reached or started, report the concrete blocker instead of switching to a
49
60
  mock target or substituting static checks for browser QA.
50
61
  4. If the requested behavior requires authentication, run
51
- `node <runner> profiles`. A missing auth config is valid and returns an empty
62
+ `node "$PI_BROWSER_QA_RUNNER" profiles`. A missing auth config returns an empty
52
63
  list without creating `.pi/qa_auth.jsonc`. Otherwise skip profile discovery
53
64
  and use public mode. Choose an auth profile only when the task names its id,
54
65
  safe profile traits make the choice unambiguous, or a public run proves that
@@ -61,7 +72,8 @@ capability probing, or retries.
61
72
  executable JavaScript in it. The `evaluate` action exposes only the safe
62
73
  operations documented below; it does not accept expressions or scripts.
63
74
  6. Run public QA with
64
- `node <runner> run --base-url <url> --flow <flow.jsonc>`. The URL's exact
75
+ `node "$PI_BROWSER_QA_RUNNER" run --base-url <url> --flow <flow.jsonc>`.
76
+ The URL's exact
65
77
  origin becomes the fail-closed allowlist. When the real app requires known
66
78
  API/CDN origins, add repeatable `--allow-origin <exact-origin>` flags. Each
67
79
  value must be an exact `http(s)` origin with no path, credentials, wildcard,
@@ -128,8 +140,8 @@ capability probing, or retries.
128
140
  product behavior differs from the expectation, preserve the failure evidence
129
141
  and report the mismatch.
130
142
 
131
- Read `references/qa-design.md` when designing a non-trivial flow, diagnosing an
132
- ambiguous failure, or deciding what evidence proves the result.
143
+ Use the detailed scenario-design guidance below for non-trivial flows,
144
+ ambiguous failures, and decisions about what evidence proves the result.
133
145
 
134
146
  ## Flow contract
135
147
 
@@ -293,7 +305,7 @@ When a public HTTPS login page (or loopback HTTP page for local development) is
293
305
  known and there is no usable profile, invoke the trusted runner once:
294
306
 
295
307
  ```sh
296
- node <runner> auth scaffold \
308
+ node "$PI_BROWSER_QA_RUNNER" auth scaffold \
297
309
  --profile <safe-id> \
298
310
  --login-url <public-login-url>
299
311
  ```
@@ -322,7 +334,7 @@ Do not request credentials merely because `.pi/qa_auth.jsonc` is absent. If the
322
334
  task explicitly requires authenticated behavior, or a public run reaches the
323
335
  flow's `authRejectedIf` check, and `profiles` returned no usable profile, run
324
336
  form-auth scaffolding when supported. Otherwise run
325
- `node <runner> profiles --require-auth`. Only these explicit authenticated paths
337
+ `node "$PI_BROWSER_QA_RUNNER" profiles --require-auth`. Only these paths
326
338
  may create the private auth file.
327
339
 
328
340
  If that command or an authenticated run returns `QA_AUTH_UPDATE_REQUIRED`, stop
@@ -345,5 +357,242 @@ These links are mandatory so the user can open the evidence directly.
345
357
  Also include `visualInspection` with `inspected` or `unavailable`; never infer a
346
358
  visual pass from a successful runner status alone.
347
359
 
348
- See `references/qa-auth.example.jsonc`, `references/qa-flow.example.jsonc`,
349
- `references/qa-design.md`, and `references/auth-scaffold-spec.md`.
360
+ ### Scaffold safety and edge cases
361
+
362
+ The trusted runner discovers the most likely form, fillable fields, submit
363
+ control, and form container on the unauthenticated page. It writes only
364
+ discovered selectors and generated placeholders to the private `0600` config.
365
+ The agent treats that config as opaque and user-owned. Scaffold status exposes
366
+ only the file path, profile id, counts, and action, never inspected DOM text,
367
+ input values, URLs, or selectors.
368
+
369
+ Scaffolding rejects pages without a discoverable fillable field or submit
370
+ control, form-less pages without an explicit success condition, remote HTTP
371
+ login pages, cross-origin base URLs, malformed success options, symlinked or
372
+ permissive auth paths, and invalid or non-empty existing configuration. Report
373
+ these blockers; do not bypass them. No generated profile is usable until the
374
+ user replaces its secret placeholders.
375
+
376
+ ## Detailed scenario-design guidance
377
+
378
+ Use the bundled declarative runner throughout. This workflow does not require
379
+ a separate browser CLI or executable Playwright scripts.
380
+
381
+ ### Build the proof before the steps
382
+
383
+ Write down three things first:
384
+
385
+ 1. **Setup:** the page and state needed to expose the behavior.
386
+ 2. **Action:** the smallest user interaction that exercises it.
387
+ 3. **Oracle:** the observable state that proves success or reproduces failure.
388
+
389
+ Good oracles are product-visible and specific: an exact URL, a stable status
390
+ message, a field value, item count, enabled/disabled state, or checked state.
391
+ Avoid treating “the click did not throw” or “the screenshot looks plausible” as
392
+ proof.
393
+
394
+ When verifying a fix, prefer a focused regression flow over a broad tour of the
395
+ application. If multiple independent states matter, assert each one explicitly.
396
+
397
+ For explicit exploratory/manual QA, keep exploration bounded rather than turning
398
+ it into an open-ended crawl. Run at most three minimal rounds. Each round should
399
+ start from one concrete hypothesis, produce a deterministic observation plus a
400
+ meaningful screenshot, and use that evidence to decide whether another round is
401
+ justified. Verification tasks remain one browser run per profile.
402
+
403
+ ### Choose resilient locators
404
+
405
+ Prefer locators that match how users and accessibility APIs identify controls:
406
+
407
+ 1. `testId` when the product exposes a stable test contract.
408
+ 2. `role` with accessible `name` for buttons, links, headings, dialogs, and
409
+ similar semantic elements.
410
+ 3. `label` for form controls.
411
+ 4. `placeholder` or visible `text` when they are stable product copy.
412
+ 5. `css` only when no semantic contract exists.
413
+
414
+ Use `exact: true` when duplicate or substring matches are possible. Avoid CSS
415
+ that encodes DOM depth, generated classes, styling details, or element order.
416
+ If a locator is ambiguous, inspect nearby source or rendered copy and choose a
417
+ more specific product contract rather than adding arbitrary delays.
418
+
419
+ ### Wait for state, not time
420
+
421
+ Runner interactions inherit Playwright auto-waiting. Usually an action followed
422
+ by an assertion is enough. Use `waitFor` only when the next operation depends on
423
+ a distinct attached/detached/visible/hidden transition.
424
+
425
+ After navigation and visible interactions, the runner also waits for DOM
426
+ readiness, tracks requests causally started by that action through a bounded
427
+ readiness window, waits for common visible
428
+ `aria-busy`/progress/loading/spinner/skeleton markers, and keeps a 500 ms stable
429
+ interval. EventSource/WebSocket traffic is ignored, and a long poll is not
430
+ allowed to pin the entire flow timeout. A visible busy indicator may still hold
431
+ readiness until `timeoutMs`. This is a safe baseline, not an application-specific
432
+ oracle: explicitly wait for a custom loader to become hidden and assert the
433
+ loaded content when the application uses different readiness semantics.
434
+
435
+ `waitForTimeout` is bounded to five seconds and should be exceptional—for a
436
+ known animation, debounce, or externally scheduled transition with no
437
+ observable intermediate state. Sleeping longer hides races instead of proving
438
+ behavior. If a normal operation legitimately needs more time, adjust the flow's
439
+ `timeoutMs` rather than inserting repeated sleeps.
440
+
441
+ All assertion actions retry until that timeout. This makes an action followed
442
+ directly by `assertText`, `assertVisible`, `assertURL`, or another assertion
443
+ safe for asynchronously rendered outcomes. `assertText` requires the locator to
444
+ be visible and matches rendered `innerText`; use `assertTextContent` only when
445
+ hidden/raw DOM text is deliberately part of the oracle. Use `assertAttribute` for
446
+ observable state such as `aria-expanded`, `aria-invalid`, or `data-state`
447
+ instead of reading DOM state through executable JavaScript.
448
+
449
+ ### Responsive and scrolling scenarios
450
+
451
+ Set the flow's top-level `viewport` whenever the bug depends on a breakpoint or
452
+ available height. Assert `viewportWidth` or `viewportHeight` with
453
+ `assertDOMMetric` when the dimensions themselves are part of the proof; the
454
+ runner also includes the applied viewport in its result.
455
+
456
+ Use `wheel` to reproduce real pointer-wheel input. Add a locator when the wheel
457
+ must target a nested scrolling container—the runner hovers it before sending
458
+ the input. Because browser scrolling may be scheduled after the wheel event,
459
+ wait only for a short known settling interval when a direct metric assertion is
460
+ otherwise racy.
461
+
462
+ Use safe `evaluate` `scrollTo`/`scrollBy` operations for deterministic setup or
463
+ to distinguish input handling from layout behavior. Use the `metrics` operation
464
+ to retain a named page/element snapshot in result `observations`, and use
465
+ `assertDOMMetric` for pass/fail. Raw JavaScript expressions are intentionally
466
+ excluded: flows remain declarative and cannot inspect authentication storage or
467
+ execute arbitrary same-origin requests.
468
+
469
+ ### Deterministic browser environment
470
+
471
+ Set top-level `environment` when locale, timezone, color scheme, or motion
472
+ preferences can change the behavior. The runner otherwise uses stable defaults
473
+ (`en-US`, `UTC`, `light`, `reduce`) instead of inheriting host settings. Prefer
474
+ asserting product-visible copy or state derived from those settings; do not use
475
+ screenshot pixels as the only oracle.
476
+
477
+ ### Causal network and dialog expectations
478
+
479
+ Put `expectResponse` or `expectDialog` on the interaction that causes the event.
480
+ The runner arms both listeners before the interaction, avoiding the race in a
481
+ separate “click, then wait” sequence. Response expectations deliberately match
482
+ only an exact allowlisted-origin pathname, HTTP method, and status. This proves
483
+ that a matching request started and received a response within the action
484
+ window without retaining its URL query, headers, or body.
485
+
486
+ Dialog expectations match a fixed dialog type and exact/included message, then
487
+ accept or dismiss it declaratively. A mismatch is dismissed before the step
488
+ fails so the page cannot freeze. Actual event metadata is never included in
489
+ failure diagnostics. Do not place secrets in expected messages or response
490
+ paths even though the runner keeps diagnostics generic.
491
+
492
+ ### Drag, upload, and download scenarios
493
+
494
+ Use `dragTo` for native DOM drag/drop and assert the resulting product state.
495
+ Optional source/drop positions are relative bounded coordinates. Canvas-only,
496
+ OS-native, or custom synthetic-event drag protocols remain unsupported; do not
497
+ work around that with executable JavaScript.
498
+
499
+ Uploads are memory-only base64 payloads declared in the flow. This intentionally
500
+ prevents a flow from selecting arbitrary project files, credential config, or
501
+ directories. Keep fixtures minimal and non-secret. An empty file list clears a
502
+ file input.
503
+
504
+ Use the atomic `download` action rather than clicking a download link directly.
505
+ Always match the suggested filename and choose a tight `maxBytes`. Retain a
506
+ download only when its contents are needed as evidence; otherwise the runner
507
+ deletes it after validation. Retained downloads use generated private names,
508
+ not server-provided paths, and are scanned for configured authentication before
509
+ publication. `maxBytes` bounds the runner's private evidence copy and triggers
510
+ cancellation while it grows, but it is not a network-bandwidth guarantee: the
511
+ browser can receive temporary bytes before cancellation.
512
+
513
+ ### Same-origin frames and popups
514
+
515
+ Use a scoped `target` for iframe or named popup interactions. The runner checks
516
+ the live frame origin before every scoped step and checks a popup after it loads;
517
+ both must remain in `allowedOrigins`. This supports embedded application UI and
518
+ same-origin auxiliary windows without opening a route around the network guard.
519
+ Cross-origin login, payment, and third-party widgets remain intentionally out of
520
+ scope. Each popup adds a separate private video artifact, so open only the
521
+ windows needed for the proof.
522
+
523
+ ### Authentication transitions
524
+
525
+ Add `authRejectedIf` directly after initial navigation and after transitions
526
+ that can redirect to login or display an expired-session marker. This converts
527
+ stale credentials into an explicit update request instead of misreporting a
528
+ product regression.
529
+
530
+ Do not encode credentials, tokens, storage values, or login form secrets in the
531
+ flow. The trusted runner applies the selected profile internally. For form auth,
532
+ video starts on the login page and includes field filling and submission; password
533
+ inputs remain masked, but visible identifiers can appear, so treat the video as
534
+ sensitive private evidence. Tracing starts only after login succeeds and is
535
+ sanitized before retention.
536
+
537
+ ### Evidence strategy
538
+
539
+ The runner always attempts a final or failure screenshot, records video from the
540
+ first page, and creates a sanitized post-auth trace. Add named `screenshot`
541
+ steps only at states that materially help explain the result—for example before
542
+ and after a destructive interaction, or when a transient success message is
543
+ the oracle.
544
+
545
+ Use evidence by purpose:
546
+
547
+ - **Screenshot:** quick review of one meaningful visual state.
548
+ - **Video:** chronological confirmation of the complete user flow.
549
+ - **Trace:** action/DOM timing diagnosis for a failed or flaky interaction.
550
+
551
+ Videos automatically show a transient cursor and yellow pulse for clicks and
552
+ double-clicks. After a native `dragTo` gesture completes, its resolved
553
+ source-to-target route is replayed over 450 ms with a large orange cursor and
554
+ progressively drawn high-contrast trail, followed by a green drop marker. These
555
+ annotations are
556
+ runner-owned, pointer-transparent, and accessibility-hidden; they cover the main
557
+ page, same-origin frames, declared popups, and form-auth submission. Their
558
+ bounded animations finish within the normal post-action stable interval, so
559
+ they explain the chronology without becoming screenshot or assertion oracles.
560
+
561
+ Assertions determine pass/fail; evidence explains it. Preserve and link every
562
+ artifact group returned on both passed and failed runs.
563
+
564
+ For visual QA, do not stop at artifact generation. Open at least one meaningful
565
+ PNG with the model's image-capable `read` path and inspect layout, clipping,
566
+ overlap, state styling, and other visual defects relevant to the scenario. If
567
+ the active model cannot read images, explicitly report visual inspection as
568
+ unavailable rather than treating deterministic assertions as a visual pass.
569
+
570
+ ### Diagnose failures without weakening the test
571
+
572
+ Classify the first failing step:
573
+
574
+ - wrong target/setup or service unavailable;
575
+ - authentication rejected or expired;
576
+ - locator no longer matches the product contract;
577
+ - expected state never appeared;
578
+ - actual product behavior contradicts the expectation.
579
+
580
+ Fix the flow only when its setup or locator is wrong. Do not replace a precise
581
+ assertion with a vague one, increase timeouts reflexively, or remove the failing
582
+ step to manufacture a pass. Keep the failure artifacts and state the expected
583
+ versus observed behavior.
584
+
585
+ ### Cleanup and isolation
586
+
587
+ Each runner invocation owns one isolated context and evidence directory and
588
+ closes its browser resources in a `finally` path. Do not create parallel shared
589
+ or default sessions outside the runner. Test multiple auth profiles with
590
+ separate invocations so cookies, storage, traces, and evidence cannot mix.
591
+
592
+ Keep the declarative flow and every generated screenshot, video, trace, and
593
+ result manifest inside `$PI_SUBAGENT_AGENT_DIR/browser-qa/`. The launcher owns
594
+ that path and the runner validates it before opening a browser. Do not override
595
+ the environment path or copy evidence into shared `.pi/qa-runs`/`.pi/qa-flows`
596
+ directories: the agent-local workspace is intentionally removed by the normal
597
+ sub-agent shutdown and cleanup lifecycle. Authentication config remains a
598
+ separate persistent input under project `.pi/`.
@@ -0,0 +1,20 @@
1
+ ---
2
+ description: Make bounded changes to code, docs, tests, or UI using the parent's requirements and acceptance criteria. Follow local conventions and verify the change. Task-specific discipline belongs in the brief, not a separate role.
3
+ icon: code
4
+ models: [zai/glm-5.3-flash, openai-codex/gpt-5.6-terra, openai-codex/gpt-5.6-luna]
5
+ thinking: medium
6
+ ---
7
+
8
+ # Implement
9
+
10
+ Execute the bounded change, including documentation, tests, or frontend work
11
+ when requested. Inspect nearby conventions, preserve unrelated work, and do
12
+ not broaden scope or make product/architecture decisions for the parent.
13
+
14
+ For UI work, preserve the existing design language and inspect supplied visual
15
+ references with an image-capable model. Report unavailable capabilities instead
16
+ of claiming visual verification. Actual browser QA belongs to browser-qa.
17
+
18
+ Run relevant targeted checks and report changed paths plus their results. If
19
+ requirements conflict or the task exceeds your capabilities, stop with a
20
+ concrete blocker and the evidence already gathered; do not re-plan the project.
@@ -0,0 +1,16 @@
1
+ ---
2
+ description: Strong independent second opinion for hard or high-stakes uncertainty. Use sparingly to challenge architecture, plans, root-cause hypotheses, or risk decisions. Prefer another provider within the available pool; read-only advice, not routine execution.
3
+ icon: sparkles
4
+ models: [openai-codex/gpt-5.6-sol, zai/glm-5.3]
5
+ thinking: max
6
+ tools: [read, grep, bash]
7
+ ---
8
+
9
+ # Oracle agent
10
+
11
+ You are an oracle: a strong model giving an independent second opinion to the
12
+ parent agent. A different provider is preferred within the available pool, but
13
+ not guaranteed; do not claim cross-provider independence without checking.
14
+ Give a concise, decisive recommendation with key
15
+ tradeoffs and risks. Disagree when warranted; do not rubber-stamp. Do not edit
16
+ unless explicitly asked.
@@ -0,0 +1,18 @@
1
+ ---
2
+ description: Read-only evidence gathering - search files, trace behavior, investigate hypotheses, or independently review a diff. Return findings with paths, not raw source. Use verify for running checks and implement for changes.
3
+ icon: search
4
+ models: [zai/glm-5-turbo, openai-codex/gpt-5.6-luna]
5
+ thinking: low
6
+ tools: [read, grep]
7
+ ---
8
+
9
+ # Research
10
+
11
+ Investigate the assigned question within scope. Read and compare evidence; do
12
+ not edit files or invent missing facts. A review is a fresh investigation of
13
+ the supplied diff, not approval based on the implementer's summary.
14
+
15
+ Return the answer and supporting file:line references, distinguishing confirmed
16
+ findings from hypotheses. Report gaps or a concrete blocker when the available
17
+ evidence is insufficient. Leave architecture decisions and escalation to the
18
+ parent. Do not dump search results or source files unless explicitly requested.
@@ -0,0 +1,18 @@
1
+ ---
2
+ description: Run targeted tests, builds, or checks and diagnose their output in an isolated context. Return pass/fail, relevant failure evidence and log paths. Do not fix source code; creating or changing tests belongs to implement.
3
+ icon: flask
4
+ models: [zai/glm-5-turbo, openai-codex/gpt-5.6-luna]
5
+ thinking: low
6
+ tools: [read, grep, bash]
7
+ ---
8
+
9
+ # Verify
10
+
11
+ Select and run the smallest checks that establish the requested acceptance
12
+ criteria. Do not edit source or tests, weaken assertions, install dependencies,
13
+ or update snapshots to manufacture a pass. Report missing prerequisites.
14
+
15
+ Keep noisy output in artifacts and return the command, exit status, relevant
16
+ failure excerpt, and log path. Separate environment failures from product
17
+ failures and mark unrun checks explicitly. This is a behavioral no-edit
18
+ contract: the shell is not a read-only filesystem sandbox.