pi-ui-extend 1.0.41 → 1.0.45

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. package/README.md +50 -13
  2. package/dist/app/app.d.ts +13 -0
  3. package/dist/app/app.js +158 -8
  4. package/dist/app/cli/install.js +2 -2
  5. package/dist/app/commands/command-controller.d.ts +4 -0
  6. package/dist/app/commands/command-controller.js +11 -0
  7. package/dist/app/commands/command-git-actions.d.ts +25 -0
  8. package/dist/app/commands/command-git-actions.js +381 -0
  9. package/dist/app/commands/command-host.d.ts +4 -0
  10. package/dist/app/commands/command-host.js +5 -2
  11. package/dist/app/commands/command-model-actions.d.ts +1 -0
  12. package/dist/app/commands/command-model-actions.js +57 -35
  13. package/dist/app/commands/command-navigation-actions.d.ts +1 -0
  14. package/dist/app/commands/command-navigation-actions.js +16 -3
  15. package/dist/app/commands/command-registry.d.ts +2 -0
  16. package/dist/app/commands/command-registry.js +15 -1
  17. package/dist/app/commands/command-session-actions.js +9 -3
  18. package/dist/app/commands/reload-context-inventory.d.ts +11 -0
  19. package/dist/app/commands/reload-context-inventory.js +70 -0
  20. package/dist/app/extensions/extension-actions-controller.d.ts +1 -0
  21. package/dist/app/extensions/extension-actions-controller.js +6 -0
  22. package/dist/app/extensions/subagent-catalog-state.d.ts +9 -0
  23. package/dist/app/extensions/subagent-catalog-state.js +23 -0
  24. package/dist/app/input/autocomplete-controller.js +61 -34
  25. package/dist/app/input/input-action-controller.d.ts +2 -0
  26. package/dist/app/input/input-action-controller.js +8 -1
  27. package/dist/app/input/input-controller.d.ts +2 -1
  28. package/dist/app/input/input-controller.js +7 -2
  29. package/dist/app/input/prompt-enhancer-controller.js +53 -35
  30. package/dist/app/input/voice-controller.d.ts +41 -45
  31. package/dist/app/input/voice-controller.js +351 -419
  32. package/dist/app/model/model-usage-controller.js +17 -7
  33. package/dist/app/model/model-usage-status.d.ts +4 -1
  34. package/dist/app/model/model-usage-status.js +270 -46
  35. package/dist/app/popup/menu-items-controller.d.ts +13 -3
  36. package/dist/app/popup/menu-items-controller.js +37 -21
  37. package/dist/app/popup/popup-action-controller.d.ts +12 -2
  38. package/dist/app/popup/popup-action-controller.js +77 -24
  39. package/dist/app/popup/popup-menu-controller.d.ts +36 -14
  40. package/dist/app/popup/popup-menu-controller.js +239 -69
  41. package/dist/app/rendering/dcp-stats.d.ts +9 -4
  42. package/dist/app/rendering/dcp-stats.js +40 -425
  43. package/dist/app/rendering/editor-panels.js +10 -4
  44. package/dist/app/rendering/popup-menu-renderer.d.ts +3 -5
  45. package/dist/app/rendering/popup-menu-renderer.js +42 -37
  46. package/dist/app/rendering/render-controller.js +23 -2
  47. package/dist/app/rendering/status-line-renderer.d.ts +5 -0
  48. package/dist/app/rendering/status-line-renderer.js +41 -5
  49. package/dist/app/rendering/tab-line-renderer.js +26 -20
  50. package/dist/app/runtime.d.ts +12 -1
  51. package/dist/app/runtime.js +120 -12
  52. package/dist/app/screen/mouse-controller.d.ts +2 -0
  53. package/dist/app/screen/mouse-controller.js +17 -7
  54. package/dist/app/screen/status-controller.d.ts +4 -0
  55. package/dist/app/screen/status-controller.js +5 -0
  56. package/dist/app/session/lazy-session-manager.js +34 -0
  57. package/dist/app/session/session-event-controller.d.ts +1 -0
  58. package/dist/app/session/session-event-controller.js +10 -1
  59. package/dist/app/session/session-history.d.ts +1 -0
  60. package/dist/app/session/session-history.js +12 -1
  61. package/dist/app/session/session-lifecycle-controller.d.ts +7 -1
  62. package/dist/app/session/session-lifecycle-controller.js +16 -1
  63. package/dist/app/session/tabs-controller.d.ts +26 -1
  64. package/dist/app/session/tabs-controller.js +454 -147
  65. package/dist/app/subagents/subagents-files.js +60 -1
  66. package/dist/app/subagents/subagents-model.d.ts +1 -0
  67. package/dist/app/subagents/subagents-model.js +18 -2
  68. package/dist/app/subagents/subagents-widget-controller.d.ts +1 -0
  69. package/dist/app/subagents/subagents-widget-controller.js +6 -0
  70. package/dist/app/types.d.ts +16 -1
  71. package/dist/app/workspace/workspace-actions-controller.js +10 -2
  72. package/dist/app/workspace/workspace-undo.d.ts +1 -0
  73. package/dist/app/workspace/workspace-undo.js +1 -0
  74. package/dist/bundled-extensions/question/index.js +9 -1
  75. package/dist/bundled-extensions/question/remote.d.ts +4 -0
  76. package/dist/bundled-extensions/question/remote.js +33 -0
  77. package/dist/bundled-extensions/telegram-connector/bot.d.ts +43 -0
  78. package/dist/bundled-extensions/telegram-connector/bot.js +166 -0
  79. package/dist/bundled-extensions/telegram-connector/config.d.ts +8 -0
  80. package/dist/bundled-extensions/telegram-connector/config.js +87 -0
  81. package/dist/bundled-extensions/telegram-connector/coordinator.d.ts +66 -0
  82. package/dist/bundled-extensions/telegram-connector/coordinator.js +413 -0
  83. package/dist/bundled-extensions/telegram-connector/index.d.ts +3 -0
  84. package/dist/bundled-extensions/telegram-connector/index.js +195 -0
  85. package/dist/bundled-extensions/terminal-bell/index.d.ts +0 -8
  86. package/dist/bundled-extensions/terminal-bell/index.js +0 -76
  87. package/dist/bundled-extensions/workspace-undo/index.d.ts +26 -0
  88. package/dist/bundled-extensions/workspace-undo/index.js +191 -0
  89. package/dist/config.d.ts +21 -2
  90. package/dist/config.js +159 -32
  91. package/dist/default-pix-config.js +25 -5
  92. package/dist/schemas/index.d.ts +1 -0
  93. package/dist/schemas/index.js +1 -0
  94. package/dist/schemas/pi-tools-suite-schema.d.ts +88 -62
  95. package/dist/schemas/pi-tools-suite-schema.js +54 -83
  96. package/dist/schemas/pix-schema.d.ts +19 -2
  97. package/dist/schemas/pix-schema.js +47 -5
  98. package/dist/schemas/tasks-schema.d.ts +18 -0
  99. package/dist/schemas/tasks-schema.js +41 -0
  100. package/docs/concurrency.md +9 -1
  101. package/docs/desktop-mvp.md +22 -6
  102. package/docs/desktop-task-manager.md +148 -84
  103. package/docs/release.md +30 -6
  104. package/external/pi-tools-suite/README.md +197 -84
  105. package/external/pi-tools-suite/docs/evals.md +1 -1
  106. package/external/pi-tools-suite/docs/session-recovery.md +47 -14
  107. package/external/pi-tools-suite/docs/subagent-model-pools.md +48 -30
  108. package/external/pi-tools-suite/docs/ui-qa-subagent.md +440 -0
  109. package/external/pi-tools-suite/package.json +6 -1
  110. package/external/pi-tools-suite/src/antigravity-auth/auth-store.ts +2 -1
  111. package/external/pi-tools-suite/src/antigravity-auth/constants.ts +9 -3
  112. package/external/pi-tools-suite/src/antigravity-auth/headers.ts +41 -3
  113. package/external/pi-tools-suite/src/antigravity-auth/models.ts +94 -21
  114. package/external/pi-tools-suite/src/antigravity-auth/oauth.ts +5 -17
  115. package/external/pi-tools-suite/src/antigravity-auth/payload.ts +72 -7
  116. package/external/pi-tools-suite/src/antigravity-auth/stream.ts +13 -1
  117. package/external/pi-tools-suite/src/async-subagents/agents/frontier-review.md +23 -0
  118. package/external/pi-tools-suite/src/async-subagents/agents/implement.md +1 -1
  119. package/external/pi-tools-suite/src/async-subagents/agents/presets.jsonc +16 -0
  120. package/external/pi-tools-suite/src/async-subagents/agents/research.md +5 -3
  121. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/backends/browser.mjs +346 -0
  122. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/backends/desktop.mjs +1038 -0
  123. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/backends/tui.mjs +759 -0
  124. package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/scripts/browser-qa-runner.mjs +32 -23
  125. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/chrome-devtools/chrome-devtools-provider.mjs +895 -0
  126. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/linux/linux-atspi.py +494 -0
  127. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/macos/macos-accessibility.swift +1085 -0
  128. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/native-terminal/bridge-client.mjs +50 -0
  129. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/native-terminal/native-terminal-host.mjs +501 -0
  130. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/windows/windows-uia.ps1 +454 -0
  131. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/browser-auth.md +114 -0
  132. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/browser-chrome-devtools.md +91 -0
  133. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/browser-playwright.md +145 -0
  134. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/browser.md +82 -0
  135. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/desktop-linux.md +26 -0
  136. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/desktop-macos.md +29 -0
  137. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/desktop-windows.md +23 -0
  138. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/desktop.md +54 -0
  139. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/tui-native-terminal.md +59 -0
  140. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/tui-pty.md +37 -0
  141. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/tui.md +60 -0
  142. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/scripts/ui-qa-runner.mjs +535 -0
  143. package/external/pi-tools-suite/src/async-subagents/agents/ui-qa.md +75 -0
  144. package/external/pi-tools-suite/src/async-subagents/commands.ts +15 -71
  145. package/external/pi-tools-suite/src/async-subagents/core/activity.ts +33 -0
  146. package/external/pi-tools-suite/src/async-subagents/core/agent-catalog.ts +5 -4
  147. package/external/pi-tools-suite/src/async-subagents/core/agent-strategy.ts +1 -1
  148. package/external/pi-tools-suite/src/async-subagents/core/agents-dir.ts +8 -5
  149. package/external/pi-tools-suite/src/async-subagents/core/browser-qa.ts +16 -2
  150. package/external/pi-tools-suite/src/async-subagents/core/config.ts +160 -327
  151. package/external/pi-tools-suite/src/async-subagents/core/model-selection.ts +2 -1
  152. package/external/pi-tools-suite/src/async-subagents/core/registry.ts +158 -33
  153. package/external/pi-tools-suite/src/async-subagents/core/routing.ts +21 -5
  154. package/external/pi-tools-suite/src/async-subagents/core/spawn.ts +40 -26
  155. package/external/pi-tools-suite/src/async-subagents/core/state.ts +40 -0
  156. package/external/pi-tools-suite/src/async-subagents/core/types.ts +7 -0
  157. package/external/pi-tools-suite/src/async-subagents/core/ultrawork-auto.ts +51 -45
  158. package/external/pi-tools-suite/src/async-subagents/index.ts +69 -5
  159. package/external/pi-tools-suite/src/async-subagents/lib.ts +16 -9
  160. package/external/pi-tools-suite/src/async-subagents/tools/spawn.ts +8 -5
  161. package/external/pi-tools-suite/src/async-subagents/tools/subagents.ts +5 -4
  162. package/external/pi-tools-suite/src/async-subagents/types.ts +1 -0
  163. package/external/pi-tools-suite/src/coding-discipline/index.ts +90 -59
  164. package/external/pi-tools-suite/src/config.ts +55 -1
  165. package/external/pi-tools-suite/src/context-gateway/accounting-log.ts +282 -0
  166. package/external/pi-tools-suite/src/context-gateway/config.ts +77 -3
  167. package/external/pi-tools-suite/src/context-gateway/efficiency.ts +414 -0
  168. package/external/pi-tools-suite/src/context-gateway/enforcement.ts +222 -0
  169. package/external/pi-tools-suite/src/context-gateway/index.ts +206 -19
  170. package/external/pi-tools-suite/src/context-gateway/storeless-capabilities.ts +7 -7
  171. package/external/pi-tools-suite/src/context-gateway/telemetry.ts +101 -28
  172. package/external/pi-tools-suite/src/context-gateway/types.ts +16 -3
  173. package/external/pi-tools-suite/src/context-inventory.ts +99 -0
  174. package/external/pi-tools-suite/src/dcp/auto-compress-budget.ts +39 -4
  175. package/external/pi-tools-suite/src/dcp/auto-compress.ts +128 -49
  176. package/external/pi-tools-suite/src/dcp/commands.ts +32 -156
  177. package/external/pi-tools-suite/src/dcp/compress-tool.ts +177 -46
  178. package/external/pi-tools-suite/src/dcp/compression-blocks.ts +6 -58
  179. package/external/pi-tools-suite/src/dcp/compression-preview.ts +9 -0
  180. package/external/pi-tools-suite/src/dcp/compression-progress.ts +22 -0
  181. package/external/pi-tools-suite/src/dcp/config.ts +114 -17
  182. package/external/pi-tools-suite/src/dcp/conversation-index.ts +36 -9
  183. package/external/pi-tools-suite/src/dcp/diagnostics.ts +41 -0
  184. package/external/pi-tools-suite/src/dcp/fresh-tool-results.ts +38 -0
  185. package/external/pi-tools-suite/src/dcp/index.ts +338 -66
  186. package/external/pi-tools-suite/src/dcp/journal.ts +126 -4
  187. package/external/pi-tools-suite/src/dcp/progress-controller.ts +2 -1
  188. package/external/pi-tools-suite/src/dcp/prompts.ts +83 -192
  189. package/external/pi-tools-suite/src/dcp/protected-continuity.ts +175 -0
  190. package/external/pi-tools-suite/src/dcp/pruner-candidates.ts +93 -3
  191. package/external/pi-tools-suite/src/dcp/pruner-message-ids.ts +6 -10
  192. package/external/pi-tools-suite/src/dcp/pruner-nudge.ts +36 -33
  193. package/external/pi-tools-suite/src/dcp/pruner-tools.ts +6 -3
  194. package/external/pi-tools-suite/src/dcp/pruner.ts +1 -0
  195. package/external/pi-tools-suite/src/dcp/routine-pressure.ts +86 -0
  196. package/external/pi-tools-suite/src/dcp/state.ts +9 -0
  197. package/external/pi-tools-suite/src/dcp/statistics.d.ts +8 -0
  198. package/external/pi-tools-suite/src/dcp/statistics.js +156 -0
  199. package/external/pi-tools-suite/src/default-pi-tools-suite-config.ts +78 -22
  200. package/external/pi-tools-suite/src/index.ts +7 -1
  201. package/external/pi-tools-suite/src/lib/project.ts +36 -1
  202. package/external/pi-tools-suite/src/model-tools/index.ts +10 -7
  203. package/external/pi-tools-suite/src/repo-discovery/index.ts +304 -4
  204. package/external/pi-tools-suite/src/resource-registry/index.ts +2551 -0
  205. package/external/pi-tools-suite/src/session-recovery/index.ts +17 -0
  206. package/external/pi-tools-suite/src/shell-command-policy.ts +219 -0
  207. package/external/pi-tools-suite/src/todo/index.ts +21 -0
  208. package/external/pi-tools-suite/src/todo/todo.ts +19 -1
  209. package/external/pi-tools-suite/src/tool-descriptions.ts +47 -19
  210. package/package.json +10 -9
  211. package/schemas/pi-tools-suite.json +466 -287
  212. package/schemas/pix.json +129 -11
  213. package/schemas/tasks.json +131 -0
  214. package/skills/simplify/SKILL.md +33 -5
  215. package/docs/desktop-markdown-media.md +0 -77
  216. package/external/pi-tools-suite/docs/browser-qa-subagent.md +0 -177
  217. package/external/pi-tools-suite/docs/dcp-emergency-current-turn.md +0 -102
  218. package/external/pi-tools-suite/src/async-subagents/agents/browser-qa.md +0 -598
  219. package/external/pi-tools-suite/src/async-subagents/async-subagents.sample.jsonc +0 -54
  220. package/external/pi-tools-suite/src/skill-installer/index.ts +0 -333
  221. package/skills/playwright-cli/SKILL.md +0 -420
  222. package/skills/playwright-cli/references/element-attributes.md +0 -23
  223. package/skills/playwright-cli/references/playwright-tests.md +0 -50
  224. package/skills/playwright-cli/references/request-mocking.md +0 -87
  225. package/skills/playwright-cli/references/running-code.md +0 -241
  226. package/skills/playwright-cli/references/session-management.md +0 -273
  227. package/skills/playwright-cli/references/spec-driven-testing.md +0 -311
  228. package/skills/playwright-cli/references/storage-state.md +0 -290
  229. package/skills/playwright-cli/references/test-generation.md +0 -142
  230. package/skills/playwright-cli/references/tracing.md +0 -154
  231. package/skills/playwright-cli/references/video-recording.md +0 -147
  232. package/skills/spec-lite/SKILL.md +0 -140
  233. /package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/examples/qa-auth.example.jsonc +0 -0
  234. /package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/examples/qa-flow.example.jsonc +0 -0
  235. /package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/vendor/fflate.LICENSE +0 -0
  236. /package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/vendor/fflate.mjs +0 -0
@@ -6,12 +6,16 @@ integration, decisions and the final answer. Actual savings depend on worker
6
6
  quality, retries and how much work the parent repeats; the configuration is
7
7
  not a price oracle.
8
8
 
9
- ## Five execution modes
9
+ ## Six execution modes
10
10
 
11
- - `research`: read-only evidence gathering, searches and independent diff review.
11
+ - `research`: read-only evidence gathering, searches and focused review questions.
12
12
  - `implement`: bounded code, documentation, test and frontend changes.
13
13
  - `verify`: run checks and interpret logs, without fixing source or tests.
14
- - `browser-qa`: isolated browser workflow with assertions and visual artifacts.
14
+ - `ui-qa`: isolated real-UI workflow for browsers, terminal/TUI apps, and
15
+ desktop GUIs with deterministic assertions and inspectable evidence.
16
+ - `frontier-review`: independent post-implementation code review on a strong
17
+ model; hidden when the current parent model matches the role's availability
18
+ gate.
15
19
  - `oracle`: a deliberate strong second opinion, not automatic worker escalation.
16
20
 
17
21
  Task-specific discipline belongs in the brief or `promptAppend`. A new project
@@ -35,33 +39,31 @@ thinking: medium
35
39
  ---
36
40
  ```
37
41
 
38
- A preset contains a set of available models, not a per-agent matrix:
42
+ A preset contains a set of available models, not a per-agent matrix. Project
43
+ pools live in `<project>/.pi/agents/presets.jsonc`:
39
44
 
40
45
  ```jsonc
41
46
  {
42
- "asyncSubagents": {
43
- "presets": {
44
- "gpt": {
45
- "description": "Models available for this session",
46
- "models": [
47
- "openai-codex/gpt-5.6-luna",
48
- "openai-codex/gpt-5.6-terra",
49
- "openai-codex/gpt-5.6-sol"
50
- ]
51
- }
52
- }
47
+ "gpt": {
48
+ "description": "Models available for this project",
49
+ "models": [
50
+ "openai-codex/gpt-5.6-luna",
51
+ "openai-codex/gpt-5.6-terra",
52
+ "openai-codex/gpt-5.6-sol"
53
+ ]
53
54
  }
54
55
  }
55
56
  ```
56
57
 
57
58
  The example agent selects Terra, not Luna: the agent's order wins. Sol is
58
59
  available in the pool but absent from this worker's chain, so it cannot become
59
- an automatic implementation fallback. The oracle can declare Sol in its own
60
- chain. Model references must be exact `provider/model` values, not wildcards.
60
+ an automatic implementation fallback. The oracle and frontier-review can
61
+ declare Sol in their own chains. Model references in `models` must be exact
62
+ `provider/model` values, not wildcards.
61
63
 
62
64
  The resolver intersects the agent chain with the selected pool. Runtime
63
65
  selection then skips unregistered, unauthenticated or session-exhausted models.
64
- Tasks with images and browser QA require confirmed image support. The first
66
+ Tasks with images and UI QA require confirmed image support. The first
65
67
  eligible candidate runs; only the remaining eligible candidates are passed to
66
68
  quota fallback. An empty intersection or unavailable chain rejects the batch
67
69
  before any children or run state are created. Model selection makes no LLM
@@ -83,22 +85,37 @@ Use `/subagent-preset <name>`, `AGENTS_PRESET=<name>` or
83
85
  without a pool filter. The shipped names remain compatible with saved choices:
84
86
  `cheap` is the GLM pool, `gpt` the GPT pool, and `deep` the mixed pool. The last
85
87
  name no longer means that ordinary workers should escalate to flagship models.
88
+ Bundled definitions live beside the built-in agents in
89
+ `src/async-subagents/agents/presets.jsonc`; a project file with the same preset
90
+ name overrides that pool.
86
91
 
87
92
  Old role names are not implicit aliases. `quick`, `scan`, `review`, `deep`,
88
93
  `docs`, `frontend`, and `tests` work only when explicitly defined as ordinary
89
94
  custom/project types. This keeps the effective catalog and accepted names exact.
90
95
 
91
- Legacy `model` plus `fallbackModels` and `modelByParent` still load. New profile
92
- `models` replaces inherited legacy selection fields; an explicit legacy model
93
- override can still replace an inherited new list. Empty `models` means no
94
- candidates, not permission to inherit the parent model. Model-less project
95
- specialists must declare candidates or receive an explicit model override.
96
-
97
- Legacy preset role matrices remain readable. A preset with `models` uses only
98
- the pool contract, dropping stale legacy model/thinking/type overrides. Switching
99
- a higher-priority config layer back to a legacy preset removes the inherited
100
- pool. Configuration loading never rewrites user files; review old overrides
101
- when migrating, since explicitly saved profiles can retain expensive models.
96
+ Agent frontmatter can gate whether a role exists for the current parent model:
97
+ `forParentModels` is an optional allow-list and `notForParentModels` is an
98
+ optional deny-list; deny wins when both match. These fields accept model
99
+ patterns such as `zai/*` and affect the parent catalog, explicit role
100
+ validation, and automatic routing. They do not change which model the child
101
+ runs on; `models` / legacy model selectors still own child model selection.
102
+
103
+ Legacy `model` plus `fallbackModels` and `modelByParent` still load when they are
104
+ declared in an agent Markdown file. New profile `models` replaces inherited
105
+ legacy selection fields. Empty `models` means no candidates, not permission to
106
+ inherit the parent model. Model-less project specialists must declare candidates
107
+ or receive an explicit model override.
108
+
109
+ Legacy singular selectors are normalized with an explicit fallback array:
110
+ `model` without `fallbackModels` resolves to `fallbackModels: []`, and every
111
+ normalized `modelByParent` entry carries its own `fallbackModels` array. Modern
112
+ `models` profiles already encode the complete ordered candidate/fallback chain
113
+ in one array and are not wrapped in an additional fallback field.
114
+
115
+ The removed `asyncSubagents` section is no longer part of the public config
116
+ schema and is not read at runtime. Existing legacy files are left untouched but
117
+ have no effect. Migrate role definitions to `<project>/.pi/agents/*.md` and
118
+ custom pools to `<project>/.pi/agents/presets.jsonc`.
102
119
 
103
120
  ## Compact handoff
104
121
 
@@ -106,4 +123,5 @@ Give workers a scope, acceptance criteria and the evidence needed to start.
106
123
  Read compact results first and inspect raw artifacts selectively. One noisy
107
124
  sequential investigation can justify a worker; a command whose exit status is
108
125
  sufficient usually only needs a saved log, not another LLM. Independent review
109
- uses a fresh `research` invocation, not a separate built-in persona.
126
+ of substantive code changes uses `frontier-review` when it is present in the
127
+ current parent catalog; use `research` for focused evidence/review questions.
@@ -0,0 +1,440 @@
1
+ # UI QA sub-agent specification
2
+
3
+ ## Type
4
+
5
+ As-is
6
+
7
+ ## Lifecycle
8
+
9
+ Active implemented contract.
10
+
11
+ ## Goal
12
+
13
+ Provide a cheap, fast `ui-qa` async-subagent that reproduces user-visible bugs
14
+ and proves fixes across browser/web UI, terminal/TUI applications, and native
15
+ desktop GUIs. Browser QA selects between the existing trusted Playwright runner
16
+ for ordinary isolated E2E/auth/video/trace work and a capability-probed Chrome
17
+ DevTools CLI provider for DevTools-specific AX, console, network, performance,
18
+ Lighthouse, and memory diagnostics. Native/TUI QA uses a real PTY or
19
+ deterministic app/platform UI driver and retains inspectable captures plus
20
+ automatic bounded video evidence when the backend supports it.
21
+ Its ranked `models` list prefers `zai/glm-5.3-flash`, then
22
+ `openai-codex/gpt-5.6-luna`, filtered by the active preset's model pool and
23
+ confirmed runtime image support.
24
+
25
+ ## Inline agent workflow and skill isolation
26
+
27
+ - The bundled role is a thin common contract: `src/async-subagents/agents/ui-qa.md`
28
+ keeps only the shared invariants (real target, backend classification,
29
+ `BLOCKED` semantics, deterministic-assertion oracle, bounded execution, owned
30
+ cleanup, private evidence) plus the base-guide/probe/run invocation syntax.
31
+ It deliberately omits browser-provider, terminal-presentation,
32
+ desktop-platform, and credential implementation details. Its body becomes the
33
+ QA child's `promptAppend` through the shared agent loader. Parent and router
34
+ catalogs include only its short `description`.
35
+ - Backend-specific instructions live in the canonical resource tree under
36
+ `src/async-subagents/agents/ui-qa/guides/`. `browser.md`, `tui.md`, and
37
+ `desktop.md` are compact base routers. Detail topics are browser
38
+ `playwright`/`chrome-devtools`/`auth`, TUI `pty`/`native-terminal`, and desktop
39
+ `macos-accessibility`/`windows-uia`/`linux-at-spi`. The child first loads only
40
+ its matching base guide through
41
+ `node "$PI_UI_QA_RUNNER" guide --backend browser|tui|desktop`, then only the
42
+ routed detail topic. The command resolves bundled files from a fixed
43
+ backend-scoped allowlist, rejects unknown or cross-backend topics, unknown
44
+ options, extra arguments, and traversal, bounds guide size, and prints only
45
+ the requested document; the model never composes or reads source paths.
46
+ - The capability-first runner lives under
47
+ `src/async-subagents/agents/ui-qa/`, with browser, PTY/TUI, and platform
48
+ desktop accessibility backends for macOS, Windows, and Linux. The trusted
49
+ Playwright browser runner, vendor dependencies/licenses, and optional legacy
50
+ JSONC examples are colocated under `agents/ui-qa/browser/`. The Chrome
51
+ DevTools provider is a bundled adapter under `agents/ui-qa/drivers/` and
52
+ capability-probes an external `chrome-devtools` CLI rather than loading a
53
+ project skill. None of these assets, including the guides, is a discoverable
54
+ skill or agent role (`agents/*.md` discovery stays non-recursive and
55
+ top-level only).
56
+ - Sub-agent processes disable normal extension discovery, then always load the
57
+ suite's model-tools extension. They load the Antigravity provider extension
58
+ only when an Antigravity model is explicitly selected. The launcher appends
59
+ `--models <effective-model>` after forwarded arguments so persisted model
60
+ patterns cannot resolve unrelated providers inside the isolated child.
61
+ - Every async sub-agent launches with `--no-skills`. `--skill` and
62
+ `--skill=...` flags are removed from `extraArgs`, and role profiles have no
63
+ skill-loading field. The thin agent Markdown plus its on-demand bundled
64
+ guides are the complete role instruction source.
65
+ - Explicit legacy `browser-qa` tasks normalize to `ui-qa`; a project-local
66
+ `browser-qa.md` profile override is migrated onto the canonical `ui-qa`
67
+ profile during config loading. Installed browser resources are part of the
68
+ canonical `ui-qa` tree, while the private runtime workspace keeps its
69
+ historical `browser-qa/` name for compatibility.
70
+ - The launcher sets `PI_UI_QA_RUNNER` to the absolute capability-first runner
71
+ and `PI_BROWSER_QA_RUNNER` to its trusted browser backend, replacing inherited
72
+ values and stripping both from ordinary children. QA uses the unified runner
73
+ for backend probe/run and the browser runner directly only for credential
74
+ profile discovery or form-auth scaffolding.
75
+ - Model-only profile overrides inherit the workflow. An explicit profile
76
+ `promptAppend` replaces the body like any other agent profile; it is not an
77
+ immutable security boundary. Runtime protections remain in the runner, and
78
+ the thin prompt's guide-routing requirement does not weaken them: every
79
+ backend action still goes through the runner's fail-closed checks.
80
+
81
+ ## Browser authentication contract
82
+
83
+ - Public browser QA requires no auth profile and does not create or require
84
+ `.pi/qa_auth.jsonc`. Its explicit base URL supplies the one exact allowed
85
+ origin, and the runner still blocks every other HTTP(S)/WebSocket origin.
86
+ - Auth profiles live in project-local `.pi/qa_auth.jsonc` and are selected by
87
+ explicit id. The file must be a real project-local file with mode `0600` on
88
+ POSIX. Profile listings expose only `id`, description, and traits.
89
+ - Listing profiles when the file is absent returns an empty list without side
90
+ effects. When authenticated QA explicitly requests credentials and that file
91
+ is absent, the runner creates a private empty template and returns
92
+ `provide_credentials`.
93
+ The sub-agent must explicitly ask the user to fill the reported file and
94
+ rerun QA; it must not read or edit the credential values itself.
95
+ - Every profile requires one or more exact `allowedOrigins`. Secret-bearing auth
96
+ is applied only to those origins; all other HTTP(S)/WebSocket traffic and
97
+ service workers are blocked during QA.
98
+ - Supported auth types are `form`, `cookie`, `localStorage`, `sessionStorage`,
99
+ `bearer`, and existing Playwright `storageState`.
100
+ - The bundled runner reads secrets internally. Credentials must never be copied
101
+ into prompts, generated QA flows, shell arguments, transcripts, reports,
102
+ or QA evidence.
103
+ - Generated browser state is private cache under `.pi/qa-auth-state`. Ephemeral
104
+ flows, evidence, and result manifests are written under the owning agent's
105
+ `.pi/subagents/<run>/<agent-id>/browser-qa/` workspace. Multiple profiles use
106
+ separate browser contexts/evidence directories, and normal sub-agent shutdown
107
+ or cleanup removes the whole workspace with its run.
108
+ - Missing, rejected, or expired explicitly selected auth returns a
109
+ machine-readable update-required status naming only the profile id, config
110
+ file, and redacted reason. The parent asks the user to update the file and
111
+ reruns; there is no `/qa-auth` command.
112
+
113
+ ## Native/TUI execution contract
114
+
115
+ - The QA target must be the actual user-facing TUI or desktop application named
116
+ by the task. Repository tests, snapshots, source inspection, or a different
117
+ CLI/web surface may support discovery but cannot substitute for requested UI
118
+ execution.
119
+ - The launcher creates a private `ui-qa/` workspace under the owning agent
120
+ directory. Native/TUI transcripts, captures, screenshots, videos, and small
121
+ temporary driver artifacts stay there and are removed with the sub-agent run.
122
+ - Repeated unnamed Desktop evidence steps receive collision-free filenames
123
+ based on their action and step index. Explicit names remain available when a
124
+ stable human-readable artifact label is useful.
125
+ - Terminal/TUI verification is selected from `target.command`, with an explicit
126
+ presentation contract that is independent of project/app identity.
127
+ New QA flows choose presentation explicitly by surface category: `pty` is for
128
+ line-oriented/plain terminal/CLI programs or protocol-focused semantic tests,
129
+ while `native-terminal` is the normal presentation for structured/full-screen
130
+ TUIs. Omitted presentation remains a `pty` compatibility default for older
131
+ flows only. Native-terminal covers real-window colors, fonts/glyphs, special
132
+ symbols, wrapping, clipping, menus/focus, and pixel geometry.
133
+ The target still runs in the runner-owned PTY used for deterministic input and
134
+ semantic assertions; the runner mirrors that same raw PTY byte stream through
135
+ a private authenticated local bridge into a fresh owned native terminal host,
136
+ whose real window supplies screenshots/video. The bridge bootstrap never
137
+ contains the target argv/cwd/env. Native-terminal provider discovery is
138
+ environment/capability based, not project based: macOS prefers installed
139
+ iTerm2 then Terminal.app; Windows uses Windows Terminal; Linux prefers kitty,
140
+ then Alacritty, GNOME Terminal, Konsole, and xterm. The provider is usable for
141
+ visual QA only when the corresponding desktop driver can also correlate and
142
+ capture its real window.
143
+ Selection is based on required capabilities/evidence, never a repository-
144
+ specific heuristic. If pixel fidelity is required but native-terminal control
145
+ is unavailable, the result is `BLOCKED`; the runner does not silently
146
+ substitute a headless replay.
147
+ - The target in either TUI presentation may use its normal explicit
148
+ project/session arguments to open deterministic state before assertions.
149
+ Non-interactive stdout from another CLI path is not TUI verification.
150
+ Automated input in both presentations goes to the same owned PTY, and native-
151
+ terminal host stdin/protocol responses are bridged back to it; flows wait for
152
+ expected text or a stable frame after `sendText`/`sendKeys` before asserting
153
+ or capturing the resulting state.
154
+ - Native desktop verification is selected from `target.application`. macOS uses
155
+ the bundled Accessibility/CGWindow/ScreenCaptureKit helper, Windows uses the
156
+ bundled PowerShell/.NET UI Automation helper, and Linux uses the bundled
157
+ Python AT-SPI helper when its runtime dependencies and graphical accessibility
158
+ bus are available. Required runtime components are capability-probed; missing
159
+ dependencies/permissions or unsupported platforms return `BLOCKED`. The agent
160
+ must not install UI automation dependencies, change OS privacy/accessibility
161
+ permissions, disable sandboxing, or operate unrelated user windows.
162
+ - PTY-presentation runs automatically retain a bounded asciicast v2 replay in
163
+ `artifacts.videos`. It is generated from timestamped PTY output and resize
164
+ events, capped at 1 MiB, and is explicitly terminal-state replay rather than
165
+ pixel evidence. Native-terminal presentation instead retains real-window
166
+ screenshots and, when exact-window capture is available, a bounded MP4 from
167
+ the owned native terminal host; its asciicast/headless captures remain
168
+ diagnostic-only and are never promoted as proof of colors/glyphs/window
169
+ geometry.
170
+ - When macOS 12.3+ ScreenCaptureKit and Screen Recording permission are
171
+ available, desktop runs automatically retain a silent H.264 MP4 of only the
172
+ correlated application window. Independent-window capture scales to fill the
173
+ Retina encoder surface so the application occupies the complete video frame
174
+ rather than a top-left subset with unused canvas. Recording is capped at 30
175
+ seconds, has no display/region fallback, and is best-effort: an unavailable
176
+ video is reported as a structured observation rather than an assertion
177
+ failure.
178
+ - Windows UIA provides exact-window PNG capture through trusted Win32 APIs but
179
+ does not yet advertise exact-window video. Linux AT-SPI provides PNG capture
180
+ only when a supported screenshot producer (`gnome-screenshot` or `scrot`) is
181
+ available; keyboard input is separately gated on a working `xdotool` in the
182
+ current graphical session. Linux/Windows video remains an explicit missing
183
+ capability rather than being synthesized from a different surface.
184
+ - Pass/fail requires a deterministic product-visible oracle such as terminal
185
+ content/state, accessibility/app-driver control state, window/dialog state,
186
+ visible copy, enabled/checked/value state, or another explicit application
187
+ result. Screenshots explain the result but are not the sole oracle.
188
+ - When no safe deterministic PTY/GUI control path is available, the correct
189
+ result is `BLOCKED`; static tests are not promoted to UI QA evidence.
190
+ - Cleanup is ownership-scoped: terminate only the PTY/session/app/driver process
191
+ created by the run, never all processes with a matching application name.
192
+ POSIX desktop launch contracts correlate and clean up the complete detached
193
+ process group, so a package-manager wrapper may hand off to its GUI descendant
194
+ without making that app unreachable or leaving it running. Windows uses the
195
+ owned launcher PID as a process-tree root: UIA resolves the actual GUI
196
+ descendant before interaction, and scoped cleanup uses `taskkill /T` against
197
+ the owned launcher plus that correlated GUI root rather than an app name.
198
+
199
+ ## Unified capability-first runner contract
200
+
201
+ - One private JSONC flow under the owning agent's `ui-qa/flows/` declares
202
+ exactly one browser URL, TUI command, or desktop application target. `probe`
203
+ reports deterministic candidate capabilities and selects the matching backend;
204
+ it also returns an authoritative `selection.guide = {backend, topic}` for the
205
+ selected provider/presentation/platform detail when one exists. The child
206
+ must load/reconcile that exact topic before `run`, which then executes the
207
+ bounded flow. The child keeps the flow at mode `0600` on POSIX before either
208
+ command, matching the runner's private-path checks.
209
+ - Browser execution adapts the unified target and steps to the existing trusted
210
+ Playwright runner for ordinary E2E or to the Chrome DevTools provider for
211
+ DevTools-only capabilities. `target.browserDriver` is `auto`, `playwright`,
212
+ or `chrome-devtools`; `auto` keeps ordinary flows and every trusted auth
213
+ profile on Playwright, and selects DevTools only when the flow requests a
214
+ DevTools-only action or explicit `target.devtools` attach/start options.
215
+ - TUI execution launches only a bounded project-local/package-runtime contract
216
+ through a real PTY, models ANSI/VT alternate-screen state with a headless
217
+ terminal, and supports text, cursor, process, resize, stability, and capture
218
+ assertions.
219
+ - Desktop execution is platform-specific behind one capability contract. macOS
220
+ uses a bundled compiled Accessibility/CGWindow helper plus ScreenCaptureKit;
221
+ Windows uses bundled PowerShell/.NET UI Automation plus Win32 window capture;
222
+ Linux uses bundled Python AT-SPI plus runtime-probed screenshot/keyboard
223
+ producers. Explicit selectors and owned launch roots must still resolve the
224
+ actual GUI descendant. Unsupported platforms or missing required control
225
+ capabilities return a structured `BLOCKED` result; unavailable best-effort
226
+ video is reported without replacing deterministic assertions.
227
+ - Results normalize selection rationale, assertions, observations, and typed
228
+ artifact groups across all backends. `BLOCKED` additionally normalizes a
229
+ parent-facing `blockedHandoff` containing the selected backend/platform
230
+ driver, missing capabilities, concrete reason, remediation string,
231
+ `manualActionRequired: true`, and
232
+ `automaticRemediationAttempted: false`. This handoff is the installation/
233
+ permission/platform-repair reference for the parent; the QA child relays it
234
+ and never performs those environment changes itself. Every runner/app/helper
235
+ process has a bounded deadline and cleanup is limited to processes launched by
236
+ that run.
237
+
238
+ ## Browser backend execution contract
239
+
240
+ - A model-authored QA flow is declarative JSONC, not executable JavaScript. The
241
+ selected provider implements a bounded set of navigation, interaction,
242
+ assertion, and evidence actions. The flow never receives a Playwright context,
243
+ DevTools protocol object, arbitrary JavaScript evaluator, or credential
244
+ values.
245
+ - Browser provider selection is capability-based and project-agnostic. The
246
+ Playwright provider remains the default for ordinary repeatable E2E, trusted
247
+ authentication, downloads, frames/popups, rich locators, deterministic
248
+ locale/timezone/motion settings, automatic video, and sanitized Playwright
249
+ trace evidence. Chrome DevTools is selected for structured accessibility-tree
250
+ snapshots, console/network assertions, Lighthouse summaries, sanitized
251
+ summaries from Chrome performance traces, and sanitized heap summaries.
252
+ Explicitly forcing a
253
+ provider that cannot implement the requested action returns `BLOCKED` rather
254
+ than weakening the scenario.
255
+ - Chrome DevTools requires `chrome-devtools-mcp >= 1.9.0` on `PATH`. Probe only
256
+ checks the CLI/version; run creates a random per-run daemon `sessionId` so its
257
+ socket/PID lifecycle does not collide with a user's existing DevTools daemon.
258
+ The runner disables JavaScript evaluation, extension/PWA/experimental tool
259
+ categories, usage statistics, and CrUX lookups; enables network-header
260
+ redaction and page-id routing; restricts filesystem writes to the run evidence
261
+ directory; and passes every exact allowed origin to the DevTools network
262
+ allowlist. Raw network headers/response bodies are not retained or surfaced.
263
+ - Chrome DevTools starts an isolated browser/profile by default. Optional
264
+ `target.devtools.browserUrl` accepts only a credential-free local loopback
265
+ HTTP origin. Attached Chrome still gets a task-owned isolated page/context by
266
+ default. `reuseExistingBrowserSession: true` is allowed only with that
267
+ loopback endpoint and is reserved for an explicitly requested reuse of the
268
+ user's already-authenticated Chrome session. Cleanup closes the task-created
269
+ page and stops only the private daemon; it never stops the external Chrome or
270
+ closes unrelated tabs.
271
+ - Trusted `.pi/qa_auth.jsonc` profiles remain Playwright-only. Chrome DevTools
272
+ cannot consume a QA auth profile, and credentials may not be moved into a
273
+ DevTools browser profile, command arguments, environment, or generated flow.
274
+ A missing/old CLI produces the standard parent-ready `blockedHandoff`; the QA
275
+ child does not install or upgrade it.
276
+ - Chrome DevTools supports the common `goto`, `reload`, `click`, `doubleClick`,
277
+ `hover`, `fill`, `press`, `waitFor`, `waitForTimeout`, `assertVisible`,
278
+ `assertText`, `assertURL`, and `screenshot` subset plus
279
+ `snapshotAccessibility`, `assertNoConsoleErrors`, `assertConsole`,
280
+ `assertNetworkRequest`, `lighthouse`, `performanceTrace`, and `heapSummary`.
281
+ Its locators are intentionally limited to AX `{role,name?,exact?}` or
282
+ `{text,exact?}` and must resolve to one UID from the latest snapshot. Raw heap
283
+ snapshots and raw performance traces are temporary and deleted after bounded
284
+ sanitized summaries are retained. Top-level `environment` remains a
285
+ Playwright-only deterministic
286
+ contract; DevTools blocks it rather than silently approximating
287
+ locale/timezone/reduced-motion behavior.
288
+ - Target discovery is a bounded preflight, not an open-ended research task. The
289
+ sub-agent invokes the runner within 45 seconds or returns `BLOCKED`; it does
290
+ not spend the full launcher budget reading source or probing prerequisites.
291
+ - The launcher injects `PI_SUBAGENT_AGENT_DIR`, pre-creates private `ui-qa/` and
292
+ `browser-qa/flows/` workspaces, and clears stale UI/browser QA files when an
293
+ agent id is reused. The runner validates the directory's project/type
294
+ metadata and refuses flows outside it; the model cannot select a shared
295
+ evidence root.
296
+ - The runner owns browser lifecycle, origin checks, auth application, tracing,
297
+ screenshots, video finalization, and redacted result output. Before retaining
298
+ a trace it removes network/non-image resource entries, redacts configured and
299
+ runtime storage credentials, and verifies those values are absent.
300
+ - After every navigation or visible interaction, the runner waits for DOM
301
+ readiness, completion of requests started by the action, and disappearance of
302
+ common visible busy/spinner/skeleton markers. It requires a 500 ms stable
303
+ interval before the next action so recordings remain readable; a page that
304
+ stays busy through the flow timeout fails closed instead of being tested as a
305
+ loading shell. App-specific readiness still requires an explicit declarative
306
+ wait/assertion in the authored flow.
307
+ - Before any page is created, the runner installs a context-wide, isolated
308
+ interaction visualizer. Recorded clicks/double-clicks show a transient cursor
309
+ and pulse. Native drag/drop is replayed for 450 ms with a large orange cursor,
310
+ progressively drawn high-contrast path, and green drop marker. It also covers
311
+ same-origin frames, declared popups, and form-auth submission. The layer is
312
+ accessibility-hidden, pointer-transparent, never cancels application events,
313
+ and its bounded animations clear within the post-action stable interval.
314
+ - Success and post-launch failure results include typed artifact groups. Every
315
+ item has an absolute filesystem path and a `file:` URI; the sub-agent must
316
+ present each item as a clickable Markdown link instead of reporting only the
317
+ evidence directory.
318
+ - Success requires deterministic assertions. Visual inspection supplements,
319
+ but never replaces, explicit expected-state checks.
320
+ - Auth rejection discovered by a QA flow is reported through the
321
+ `authRejectedIf` action so the parent gets an update-required status.
322
+
323
+ ## Reliability and shutdown contract
324
+
325
+ - The built-in `ui-qa` profile has a 300-second wall-clock budget unless
326
+ the caller explicitly supplies a task or spawn timeout. This bounds model
327
+ stalls as well as browser work.
328
+ - The trusted runner has its own bounded lifecycle. Browser launch, context
329
+ setup, auth, flow execution, evidence finalization, and browser shutdown must
330
+ not wait forever; a timeout reports the last started stage without exposing
331
+ flow contents or credentials.
332
+ - Trace sanitization runs in a memory-limited worker that can be terminated at
333
+ the cleanup deadline; synchronous archive work cannot defeat the watchdog.
334
+ - The launcher always writes a small sanitized `progress.jsonl` journal in the
335
+ agent directory. It records lifecycle/RPC event types and tool names, but not
336
+ prompts, tool arguments, tool results, model text, or secrets. The browser
337
+ runner writes similarly sanitized stage entries under its private workspace.
338
+ - On POSIX, newly launched agents own a process group. Settled, timed-out, and
339
+ explicitly stopped agents signal that group rather than only the Pi process;
340
+ timeout/settled shutdown escalates to `SIGKILL` after its grace period. On
341
+ Windows the existing recursive `taskkill /T /F` behavior remains in force.
342
+ - Process-tree cleanup is scoped to a launcher-created process-group marker so
343
+ an old or externally-created PID is never treated as an owned process group.
344
+ User browser sessions outside that group must not be signalled.
345
+ - Playwright can launch Chromium in its own POSIX process group. On runner
346
+ failure the runner snapshots and kills only its own descendants before it
347
+ exits, covering that detached browser tree without touching a user's browser.
348
+
349
+ ## Related files
350
+
351
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa.md`
352
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/`
353
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/scripts/ui-qa-runner.mjs`
354
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/backends/`
355
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/chrome-devtools/chrome-devtools-provider.mjs`
356
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/native-terminal/native-terminal-host.mjs`
357
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/native-terminal/bridge-client.mjs`
358
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/macos/macos-accessibility.swift`
359
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/windows/windows-uia.ps1`
360
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/linux/linux-atspi.py`
361
+ - `external/pi-tools-suite/src/async-subagents/core/browser-qa.ts`
362
+ - `external/pi-tools-suite/src/async-subagents/core/spawn.ts`
363
+ - `external/pi-tools-suite/src/async-subagents/agents/ui-qa/browser/scripts/browser-qa-runner.mjs`
364
+ - `external/pi-tools-suite/test/async-subagents/core.test.ts`
365
+ - `external/pi-tools-suite/test/async-subagents/browser-qa-runner.test.ts`
366
+ - `external/pi-tools-suite/test/async-subagents/browser-qa-runner.e2e.test.ts`
367
+ - `external/pi-tools-suite/test/async-subagents/ui-qa-runner.test.ts`
368
+ - `external/pi-tools-suite/test/async-subagents/ui-qa-desktop.e2e.test.ts`
369
+ - `external/pi-tools-suite/test/async-subagents/selection-e2e.test.ts`
370
+
371
+ ## Acceptance criteria
372
+
373
+ 1. `ui-qa` resolves to the intended model/fallback and its inline Markdown
374
+ workflow; explicit legacy `browser-qa` requests resolve to it, and its
375
+ isolated child process can register the configured
376
+ model provider.
377
+ 2. Every child spawn contains `--no-skills` but no `--skill`. The QA child
378
+ receives the thin common contract plus guide-routing workflow in its initial
379
+ prompt, loads exactly one base backend guide and only its routed detail topic
380
+ through `PI_UI_QA_RUNNER guide`, treats `selection.guide` as authoritative
381
+ before `run`, and can reach the credential-owning browser backend through
382
+ `PI_BROWSER_QA_RUNNER` when the explicit auth topic requires it, from an
383
+ unrelated project directory.
384
+ Ordinary profiles are equally skill-free and do not receive QA-only
385
+ environment paths.
386
+ 3. Auth profile listing and all error output are redacted; model-authored input
387
+ cannot execute code in the credential-bearing process.
388
+ 4. Runner tests cover public execution without an auth file, explicit profile
389
+ selection, all auth modes, fail-closed origins, path/mode hardening, private
390
+ empty-template creation only on an explicit auth request, non-executable
391
+ flows, and successful redacted evidence creation.
392
+ 5. Native/TUI evidence, including bounded PTY replay, native-terminal real-
393
+ window screenshots/video when requested and available, and exact-window
394
+ desktop video when available, lives under the owning agent's `ui-qa/`
395
+ workspace; browser flows/evidence remain under its browser-backend
396
+ `browser-qa/` workspace. Deleting the run removes both while persistent auth
397
+ config/state remains.
398
+ 6. Runner tests prove that network activity and visible loading indicators are
399
+ awaited, persistent loading fails the flow, visible actions retain a stable
400
+ 500 ms video interval, and context-wide click/drag video visualization is
401
+ installed with bounded click pacing.
402
+ 7. Completed test runs report clickable screenshot, video, and trace links
403
+ whenever those artifacts exist.
404
+ 8. Timeout tests identify the last browser stage, launcher progress remains
405
+ available when full RPC logging is disabled, and process-tree tests prove a
406
+ descendant is terminated without signalling unrelated processes.
407
+ 9. TUI/native instructions require a real PTY/app driver, deterministic
408
+ product-visible oracles, scoped cleanup, and a `BLOCKED` result when safe
409
+ automation is unavailable rather than substituting source/unit tests.
410
+ Pixel-sensitive TUI tasks select native-terminal presentation by capability
411
+ need rather than app identity; the target still runs in one owned PTY while
412
+ a private bridge mirrors its bytes into an owned real terminal window.
413
+ 10. Suite tests/typecheck, host checks, and suite sync pass.
414
+ 11. Unified runner tests cover deterministic backend/presentation selection,
415
+ real PTY screen state and scoped cleanup, native-terminal bridge bootstrap
416
+ isolation, Windows/Linux native-host launch contracts, unsafe launch/path
417
+ rejection, timeout bounds, platform blockers, and the trusted Windows UIA/
418
+ Linux AT-SPI helper protocols. The opt-in macOS E2E launches a real AppKit
419
+ window, semantically activates its control, and retains accessibility,
420
+ screenshot, and automatic exact-window video evidence. Windows/Linux
421
+ real-host smokes are required before claiming those platform integrations
422
+ runtime-verified; macOS tests do not substitute for that evidence.
423
+
424
+ ## Real-browser regression test
425
+
426
+ The repository includes a local mock-page E2E that launches real Chromium and
427
+ asserts PNG screenshots, WebM video, sanitized trace output, and absolute
428
+ path/`file:` URI metadata:
429
+
430
+ ```bash
431
+ npx playwright install chromium
432
+ npm run test:browser-qa-e2e
433
+ ```
434
+
435
+ Normal suite tests keep this case skipped; the Publish workflow runs it on
436
+ Linux after installing Chromium. The runner writes into a temporary simulated
437
+ sub-agent directory. For manual inspection only, explicit E2E runs copy the
438
+ latest artifacts to `.pi/qa-runs/browser-qa-e2e/latest/` and print clickable
439
+ links; this test-only published copy is not the runtime storage contract. Set
440
+ `BROWSER_QA_KEEP_EVIDENCE=0` to skip that copy.
@@ -22,12 +22,14 @@
22
22
  "smoke:tools": "PI_OFFLINE=1 pi --no-session -p \"ping\"",
23
23
  "smoke": "npm run smoke:explicit && npm run smoke:auto && npm run smoke:tools",
24
24
  "test": "bun test test",
25
+ "test:dcp-session-sim": "bun test test/dcp-session-sim-e2e.test.ts",
26
+ "test:dcp-reminder-e2e": "DCP_REMINDER_E2E=1 bun test test/prompt-evals/dcp-reminder-e2e.test.ts",
25
27
  "test:browser-qa-e2e": "BROWSER_QA_RUNNER_E2E=1 bun test test/async-subagents/browser-qa-runner.e2e.test.ts",
26
28
  "test:async-subagents-e2e": "ASYNC_SUBAGENTS_E2E=1 ASYNC_SUBAGENTS_DEBUG_LOGS=1 ASYNC_SUBAGENTS_MODEL=zai/glm-5-turbo bun test --concurrent --max-concurrency=30 test/async-subagents",
27
29
  "test:async-subagents-selection-e2e": "ASYNC_SUBAGENTS_SELECTION_E2E=1 ASYNC_SUBAGENTS_MODEL=zai/glm-5-turbo bun test --concurrent --max-concurrency=30 test/async-subagents/selection-e2e.test.ts",
28
30
  "test:prompt-evals:tool-selection": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=10 test/tool-selection-e2e.test.ts",
29
31
  "test:prompt-evals:async": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/async-subagents/selection-e2e.test.ts test/prompt-evals/async-routing-e2e.test.ts",
30
- "test:prompt-evals:dcp": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/prompt-evals/dcp-summary-e2e.test.ts",
32
+ "test:prompt-evals:dcp": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/prompt-evals/dcp-summary-e2e.test.ts test/prompt-evals/dcp-reminder-e2e.test.ts",
31
33
  "test:prompt-evals": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/tool-selection-e2e.test.ts test/async-subagents/selection-e2e.test.ts test/prompt-evals",
32
34
  "test:evals:contracts": "bun test test/evals/extension-contracts.test.ts test/evals/harness.test.ts",
33
35
  "test:evals:live": "PI_TOOLS_SUITE_EVALS_LIVE=1 bun test --concurrent --max-concurrency=4 test/evals/live-evals.test.ts",
@@ -46,6 +48,9 @@
46
48
  "check": "npm run typecheck && npm test && npm run smoke"
47
49
  },
48
50
  "dependencies": {
51
+ "@cortexkit/antigravity-auth-core": "2.2.1",
52
+ "@lydell/node-pty": "1.1.0",
53
+ "@xterm/headless": "^6.0.0",
49
54
  "jsonc-parser": "^3.3.1",
50
55
  "vscode-jsonrpc": "^8.2.1",
51
56
  "vscode-languageserver-protocol": "^3.17.5"
@@ -3,6 +3,7 @@ import { promises as fs } from "node:fs";
3
3
  import { homedir } from "node:os";
4
4
  import { basename, dirname, join } from "node:path";
5
5
  import { getAgentDir } from "@earendil-works/pi-coding-agent";
6
+ import { ANTIGRAVITY_CLIENT_ID, ANTIGRAVITY_CLIENT_SECRET } from "@cortexkit/antigravity-auth-core";
6
7
  import { DEFAULT_PROJECT_ID, PROVIDER_ID } from "./constants";
7
8
  import type { GoogleOAuthClientCredentials, OpencodeAntigravityAccount, OpencodeAntigravityImportResult, OpencodeAntigravityStorage, PiAuthCredential, PiAuthData } from "./types";
8
9
 
@@ -113,7 +114,7 @@ export function getGoogleOAuthClientCredentials(...sources: Array<unknown>): Goo
113
114
  const clientId = process.env.PI_ANTIGRAVITY_GOOGLE_CLIENT_ID;
114
115
  const clientSecret = process.env.PI_ANTIGRAVITY_GOOGLE_CLIENT_SECRET;
115
116
  if (clientId) return { clientId, ...(clientSecret ? { clientSecret } : {}) };
116
- return undefined;
117
+ return { clientId: ANTIGRAVITY_CLIENT_ID, clientSecret: ANTIGRAVITY_CLIENT_SECRET };
117
118
  }
118
119
 
119
120
  export function clampAccountIndex(index: unknown, accountCount: number): number {
@@ -4,6 +4,10 @@ export const STATUS_KEY = "dcp:antigravity";
4
4
  export const LEGACY_STATUS_KEY = "antigravity";
5
5
  export const ALL_ACCOUNTS_EXHAUSTED_MARKER = "ANTIGRAVITY_ALL_ACCOUNTS_EXHAUSTED";
6
6
 
7
+ // Captured native agy CLI wire identity used by cortexkit 2.2.1.
8
+ export const AGY_CLI_VERSION = "1.1.24";
9
+ export const AGY_CLI_CHANGE_LIST = "974782877";
10
+
7
11
  export const REDIRECT_URI = "http://localhost:51121/oauth-callback";
8
12
  export const SCOPES = [
9
13
  "https://www.googleapis.com/auth/cloud-platform",
@@ -13,11 +17,13 @@ export const SCOPES = [
13
17
  "https://www.googleapis.com/auth/experimentsandconfigs",
14
18
  ];
15
19
 
16
- export const ENDPOINT_DAILY = "https://daily-cloudcode-pa.sandbox.googleapis.com";
20
+ export const ENDPOINT_DAILY = "https://daily-cloudcode-pa.googleapis.com";
17
21
  export const ENDPOINT_PROD = "https://cloudcode-pa.googleapis.com";
18
22
  export const ENDPOINT_AUTOPUSH = "https://autopush-cloudcode-pa.sandbox.googleapis.com";
19
- export const STREAM_ENDPOINTS = [ENDPOINT_DAILY, ENDPOINT_AUTOPUSH, ENDPOINT_PROD];
20
- export const LOAD_ENDPOINTS = [ENDPOINT_PROD, ENDPOINT_DAILY, ENDPOINT_AUTOPUSH];
23
+ // Autopush is retained as a named legacy endpoint for compatibility/debugging,
24
+ // but current agy/cortexkit traffic falls back daily -> prod only.
25
+ export const STREAM_ENDPOINTS = [ENDPOINT_DAILY, ENDPOINT_PROD];
26
+ export const LOAD_ENDPOINTS = [ENDPOINT_DAILY, ENDPOINT_PROD];
21
27
  export const DEFAULT_PROJECT_ID = "rising-fact-p41fc";
22
28
  export const TOKEN_EXPIRY_SKEW_MS = 5 * 60 * 1000;
23
29
  export const SKIP_THOUGHT_SIGNATURE = "skip_thought_signature_validator";