pi-ui-extend 1.0.41 → 1.0.45
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -13
- package/dist/app/app.d.ts +13 -0
- package/dist/app/app.js +158 -8
- package/dist/app/cli/install.js +2 -2
- package/dist/app/commands/command-controller.d.ts +4 -0
- package/dist/app/commands/command-controller.js +11 -0
- package/dist/app/commands/command-git-actions.d.ts +25 -0
- package/dist/app/commands/command-git-actions.js +381 -0
- package/dist/app/commands/command-host.d.ts +4 -0
- package/dist/app/commands/command-host.js +5 -2
- package/dist/app/commands/command-model-actions.d.ts +1 -0
- package/dist/app/commands/command-model-actions.js +57 -35
- package/dist/app/commands/command-navigation-actions.d.ts +1 -0
- package/dist/app/commands/command-navigation-actions.js +16 -3
- package/dist/app/commands/command-registry.d.ts +2 -0
- package/dist/app/commands/command-registry.js +15 -1
- package/dist/app/commands/command-session-actions.js +9 -3
- package/dist/app/commands/reload-context-inventory.d.ts +11 -0
- package/dist/app/commands/reload-context-inventory.js +70 -0
- package/dist/app/extensions/extension-actions-controller.d.ts +1 -0
- package/dist/app/extensions/extension-actions-controller.js +6 -0
- package/dist/app/extensions/subagent-catalog-state.d.ts +9 -0
- package/dist/app/extensions/subagent-catalog-state.js +23 -0
- package/dist/app/input/autocomplete-controller.js +61 -34
- package/dist/app/input/input-action-controller.d.ts +2 -0
- package/dist/app/input/input-action-controller.js +8 -1
- package/dist/app/input/input-controller.d.ts +2 -1
- package/dist/app/input/input-controller.js +7 -2
- package/dist/app/input/prompt-enhancer-controller.js +53 -35
- package/dist/app/input/voice-controller.d.ts +41 -45
- package/dist/app/input/voice-controller.js +351 -419
- package/dist/app/model/model-usage-controller.js +17 -7
- package/dist/app/model/model-usage-status.d.ts +4 -1
- package/dist/app/model/model-usage-status.js +270 -46
- package/dist/app/popup/menu-items-controller.d.ts +13 -3
- package/dist/app/popup/menu-items-controller.js +37 -21
- package/dist/app/popup/popup-action-controller.d.ts +12 -2
- package/dist/app/popup/popup-action-controller.js +77 -24
- package/dist/app/popup/popup-menu-controller.d.ts +36 -14
- package/dist/app/popup/popup-menu-controller.js +239 -69
- package/dist/app/rendering/dcp-stats.d.ts +9 -4
- package/dist/app/rendering/dcp-stats.js +40 -425
- package/dist/app/rendering/editor-panels.js +10 -4
- package/dist/app/rendering/popup-menu-renderer.d.ts +3 -5
- package/dist/app/rendering/popup-menu-renderer.js +42 -37
- package/dist/app/rendering/render-controller.js +23 -2
- package/dist/app/rendering/status-line-renderer.d.ts +5 -0
- package/dist/app/rendering/status-line-renderer.js +41 -5
- package/dist/app/rendering/tab-line-renderer.js +26 -20
- package/dist/app/runtime.d.ts +12 -1
- package/dist/app/runtime.js +120 -12
- package/dist/app/screen/mouse-controller.d.ts +2 -0
- package/dist/app/screen/mouse-controller.js +17 -7
- package/dist/app/screen/status-controller.d.ts +4 -0
- package/dist/app/screen/status-controller.js +5 -0
- package/dist/app/session/lazy-session-manager.js +34 -0
- package/dist/app/session/session-event-controller.d.ts +1 -0
- package/dist/app/session/session-event-controller.js +10 -1
- package/dist/app/session/session-history.d.ts +1 -0
- package/dist/app/session/session-history.js +12 -1
- package/dist/app/session/session-lifecycle-controller.d.ts +7 -1
- package/dist/app/session/session-lifecycle-controller.js +16 -1
- package/dist/app/session/tabs-controller.d.ts +26 -1
- package/dist/app/session/tabs-controller.js +454 -147
- package/dist/app/subagents/subagents-files.js +60 -1
- package/dist/app/subagents/subagents-model.d.ts +1 -0
- package/dist/app/subagents/subagents-model.js +18 -2
- package/dist/app/subagents/subagents-widget-controller.d.ts +1 -0
- package/dist/app/subagents/subagents-widget-controller.js +6 -0
- package/dist/app/types.d.ts +16 -1
- package/dist/app/workspace/workspace-actions-controller.js +10 -2
- package/dist/app/workspace/workspace-undo.d.ts +1 -0
- package/dist/app/workspace/workspace-undo.js +1 -0
- package/dist/bundled-extensions/question/index.js +9 -1
- package/dist/bundled-extensions/question/remote.d.ts +4 -0
- package/dist/bundled-extensions/question/remote.js +33 -0
- package/dist/bundled-extensions/telegram-connector/bot.d.ts +43 -0
- package/dist/bundled-extensions/telegram-connector/bot.js +166 -0
- package/dist/bundled-extensions/telegram-connector/config.d.ts +8 -0
- package/dist/bundled-extensions/telegram-connector/config.js +87 -0
- package/dist/bundled-extensions/telegram-connector/coordinator.d.ts +66 -0
- package/dist/bundled-extensions/telegram-connector/coordinator.js +413 -0
- package/dist/bundled-extensions/telegram-connector/index.d.ts +3 -0
- package/dist/bundled-extensions/telegram-connector/index.js +195 -0
- package/dist/bundled-extensions/terminal-bell/index.d.ts +0 -8
- package/dist/bundled-extensions/terminal-bell/index.js +0 -76
- package/dist/bundled-extensions/workspace-undo/index.d.ts +26 -0
- package/dist/bundled-extensions/workspace-undo/index.js +191 -0
- package/dist/config.d.ts +21 -2
- package/dist/config.js +159 -32
- package/dist/default-pix-config.js +25 -5
- package/dist/schemas/index.d.ts +1 -0
- package/dist/schemas/index.js +1 -0
- package/dist/schemas/pi-tools-suite-schema.d.ts +88 -62
- package/dist/schemas/pi-tools-suite-schema.js +54 -83
- package/dist/schemas/pix-schema.d.ts +19 -2
- package/dist/schemas/pix-schema.js +47 -5
- package/dist/schemas/tasks-schema.d.ts +18 -0
- package/dist/schemas/tasks-schema.js +41 -0
- package/docs/concurrency.md +9 -1
- package/docs/desktop-mvp.md +22 -6
- package/docs/desktop-task-manager.md +148 -84
- package/docs/release.md +30 -6
- package/external/pi-tools-suite/README.md +197 -84
- package/external/pi-tools-suite/docs/evals.md +1 -1
- package/external/pi-tools-suite/docs/session-recovery.md +47 -14
- package/external/pi-tools-suite/docs/subagent-model-pools.md +48 -30
- package/external/pi-tools-suite/docs/ui-qa-subagent.md +440 -0
- package/external/pi-tools-suite/package.json +6 -1
- package/external/pi-tools-suite/src/antigravity-auth/auth-store.ts +2 -1
- package/external/pi-tools-suite/src/antigravity-auth/constants.ts +9 -3
- package/external/pi-tools-suite/src/antigravity-auth/headers.ts +41 -3
- package/external/pi-tools-suite/src/antigravity-auth/models.ts +94 -21
- package/external/pi-tools-suite/src/antigravity-auth/oauth.ts +5 -17
- package/external/pi-tools-suite/src/antigravity-auth/payload.ts +72 -7
- package/external/pi-tools-suite/src/antigravity-auth/stream.ts +13 -1
- package/external/pi-tools-suite/src/async-subagents/agents/frontier-review.md +23 -0
- package/external/pi-tools-suite/src/async-subagents/agents/implement.md +1 -1
- package/external/pi-tools-suite/src/async-subagents/agents/presets.jsonc +16 -0
- package/external/pi-tools-suite/src/async-subagents/agents/research.md +5 -3
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/backends/browser.mjs +346 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/backends/desktop.mjs +1038 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/backends/tui.mjs +759 -0
- package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/scripts/browser-qa-runner.mjs +32 -23
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/chrome-devtools/chrome-devtools-provider.mjs +895 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/linux/linux-atspi.py +494 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/macos/macos-accessibility.swift +1085 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/native-terminal/bridge-client.mjs +50 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/native-terminal/native-terminal-host.mjs +501 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/windows/windows-uia.ps1 +454 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/browser-auth.md +114 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/browser-chrome-devtools.md +91 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/browser-playwright.md +145 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/browser.md +82 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/desktop-linux.md +26 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/desktop-macos.md +29 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/desktop-windows.md +23 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/desktop.md +54 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/tui-native-terminal.md +59 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/tui-pty.md +37 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/tui.md +60 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa/scripts/ui-qa-runner.mjs +535 -0
- package/external/pi-tools-suite/src/async-subagents/agents/ui-qa.md +75 -0
- package/external/pi-tools-suite/src/async-subagents/commands.ts +15 -71
- package/external/pi-tools-suite/src/async-subagents/core/activity.ts +33 -0
- package/external/pi-tools-suite/src/async-subagents/core/agent-catalog.ts +5 -4
- package/external/pi-tools-suite/src/async-subagents/core/agent-strategy.ts +1 -1
- package/external/pi-tools-suite/src/async-subagents/core/agents-dir.ts +8 -5
- package/external/pi-tools-suite/src/async-subagents/core/browser-qa.ts +16 -2
- package/external/pi-tools-suite/src/async-subagents/core/config.ts +160 -327
- package/external/pi-tools-suite/src/async-subagents/core/model-selection.ts +2 -1
- package/external/pi-tools-suite/src/async-subagents/core/registry.ts +158 -33
- package/external/pi-tools-suite/src/async-subagents/core/routing.ts +21 -5
- package/external/pi-tools-suite/src/async-subagents/core/spawn.ts +40 -26
- package/external/pi-tools-suite/src/async-subagents/core/state.ts +40 -0
- package/external/pi-tools-suite/src/async-subagents/core/types.ts +7 -0
- package/external/pi-tools-suite/src/async-subagents/core/ultrawork-auto.ts +51 -45
- package/external/pi-tools-suite/src/async-subagents/index.ts +69 -5
- package/external/pi-tools-suite/src/async-subagents/lib.ts +16 -9
- package/external/pi-tools-suite/src/async-subagents/tools/spawn.ts +8 -5
- package/external/pi-tools-suite/src/async-subagents/tools/subagents.ts +5 -4
- package/external/pi-tools-suite/src/async-subagents/types.ts +1 -0
- package/external/pi-tools-suite/src/coding-discipline/index.ts +90 -59
- package/external/pi-tools-suite/src/config.ts +55 -1
- package/external/pi-tools-suite/src/context-gateway/accounting-log.ts +282 -0
- package/external/pi-tools-suite/src/context-gateway/config.ts +77 -3
- package/external/pi-tools-suite/src/context-gateway/efficiency.ts +414 -0
- package/external/pi-tools-suite/src/context-gateway/enforcement.ts +222 -0
- package/external/pi-tools-suite/src/context-gateway/index.ts +206 -19
- package/external/pi-tools-suite/src/context-gateway/storeless-capabilities.ts +7 -7
- package/external/pi-tools-suite/src/context-gateway/telemetry.ts +101 -28
- package/external/pi-tools-suite/src/context-gateway/types.ts +16 -3
- package/external/pi-tools-suite/src/context-inventory.ts +99 -0
- package/external/pi-tools-suite/src/dcp/auto-compress-budget.ts +39 -4
- package/external/pi-tools-suite/src/dcp/auto-compress.ts +128 -49
- package/external/pi-tools-suite/src/dcp/commands.ts +32 -156
- package/external/pi-tools-suite/src/dcp/compress-tool.ts +177 -46
- package/external/pi-tools-suite/src/dcp/compression-blocks.ts +6 -58
- package/external/pi-tools-suite/src/dcp/compression-preview.ts +9 -0
- package/external/pi-tools-suite/src/dcp/compression-progress.ts +22 -0
- package/external/pi-tools-suite/src/dcp/config.ts +114 -17
- package/external/pi-tools-suite/src/dcp/conversation-index.ts +36 -9
- package/external/pi-tools-suite/src/dcp/diagnostics.ts +41 -0
- package/external/pi-tools-suite/src/dcp/fresh-tool-results.ts +38 -0
- package/external/pi-tools-suite/src/dcp/index.ts +338 -66
- package/external/pi-tools-suite/src/dcp/journal.ts +126 -4
- package/external/pi-tools-suite/src/dcp/progress-controller.ts +2 -1
- package/external/pi-tools-suite/src/dcp/prompts.ts +83 -192
- package/external/pi-tools-suite/src/dcp/protected-continuity.ts +175 -0
- package/external/pi-tools-suite/src/dcp/pruner-candidates.ts +93 -3
- package/external/pi-tools-suite/src/dcp/pruner-message-ids.ts +6 -10
- package/external/pi-tools-suite/src/dcp/pruner-nudge.ts +36 -33
- package/external/pi-tools-suite/src/dcp/pruner-tools.ts +6 -3
- package/external/pi-tools-suite/src/dcp/pruner.ts +1 -0
- package/external/pi-tools-suite/src/dcp/routine-pressure.ts +86 -0
- package/external/pi-tools-suite/src/dcp/state.ts +9 -0
- package/external/pi-tools-suite/src/dcp/statistics.d.ts +8 -0
- package/external/pi-tools-suite/src/dcp/statistics.js +156 -0
- package/external/pi-tools-suite/src/default-pi-tools-suite-config.ts +78 -22
- package/external/pi-tools-suite/src/index.ts +7 -1
- package/external/pi-tools-suite/src/lib/project.ts +36 -1
- package/external/pi-tools-suite/src/model-tools/index.ts +10 -7
- package/external/pi-tools-suite/src/repo-discovery/index.ts +304 -4
- package/external/pi-tools-suite/src/resource-registry/index.ts +2551 -0
- package/external/pi-tools-suite/src/session-recovery/index.ts +17 -0
- package/external/pi-tools-suite/src/shell-command-policy.ts +219 -0
- package/external/pi-tools-suite/src/todo/index.ts +21 -0
- package/external/pi-tools-suite/src/todo/todo.ts +19 -1
- package/external/pi-tools-suite/src/tool-descriptions.ts +47 -19
- package/package.json +10 -9
- package/schemas/pi-tools-suite.json +466 -287
- package/schemas/pix.json +129 -11
- package/schemas/tasks.json +131 -0
- package/skills/simplify/SKILL.md +33 -5
- package/docs/desktop-markdown-media.md +0 -77
- package/external/pi-tools-suite/docs/browser-qa-subagent.md +0 -177
- package/external/pi-tools-suite/docs/dcp-emergency-current-turn.md +0 -102
- package/external/pi-tools-suite/src/async-subagents/agents/browser-qa.md +0 -598
- package/external/pi-tools-suite/src/async-subagents/async-subagents.sample.jsonc +0 -54
- package/external/pi-tools-suite/src/skill-installer/index.ts +0 -333
- package/skills/playwright-cli/SKILL.md +0 -420
- package/skills/playwright-cli/references/element-attributes.md +0 -23
- package/skills/playwright-cli/references/playwright-tests.md +0 -50
- package/skills/playwright-cli/references/request-mocking.md +0 -87
- package/skills/playwright-cli/references/running-code.md +0 -241
- package/skills/playwright-cli/references/session-management.md +0 -273
- package/skills/playwright-cli/references/spec-driven-testing.md +0 -311
- package/skills/playwright-cli/references/storage-state.md +0 -290
- package/skills/playwright-cli/references/test-generation.md +0 -142
- package/skills/playwright-cli/references/tracing.md +0 -154
- package/skills/playwright-cli/references/video-recording.md +0 -147
- package/skills/spec-lite/SKILL.md +0 -140
- /package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/examples/qa-auth.example.jsonc +0 -0
- /package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/examples/qa-flow.example.jsonc +0 -0
- /package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/vendor/fflate.LICENSE +0 -0
- /package/external/pi-tools-suite/src/async-subagents/agents/{browser-qa → ui-qa/browser}/vendor/fflate.mjs +0 -0
|
@@ -6,12 +6,16 @@ integration, decisions and the final answer. Actual savings depend on worker
|
|
|
6
6
|
quality, retries and how much work the parent repeats; the configuration is
|
|
7
7
|
not a price oracle.
|
|
8
8
|
|
|
9
|
-
##
|
|
9
|
+
## Six execution modes
|
|
10
10
|
|
|
11
|
-
- `research`: read-only evidence gathering, searches and
|
|
11
|
+
- `research`: read-only evidence gathering, searches and focused review questions.
|
|
12
12
|
- `implement`: bounded code, documentation, test and frontend changes.
|
|
13
13
|
- `verify`: run checks and interpret logs, without fixing source or tests.
|
|
14
|
-
- `
|
|
14
|
+
- `ui-qa`: isolated real-UI workflow for browsers, terminal/TUI apps, and
|
|
15
|
+
desktop GUIs with deterministic assertions and inspectable evidence.
|
|
16
|
+
- `frontier-review`: independent post-implementation code review on a strong
|
|
17
|
+
model; hidden when the current parent model matches the role's availability
|
|
18
|
+
gate.
|
|
15
19
|
- `oracle`: a deliberate strong second opinion, not automatic worker escalation.
|
|
16
20
|
|
|
17
21
|
Task-specific discipline belongs in the brief or `promptAppend`. A new project
|
|
@@ -35,33 +39,31 @@ thinking: medium
|
|
|
35
39
|
---
|
|
36
40
|
```
|
|
37
41
|
|
|
38
|
-
A preset contains a set of available models, not a per-agent matrix
|
|
42
|
+
A preset contains a set of available models, not a per-agent matrix. Project
|
|
43
|
+
pools live in `<project>/.pi/agents/presets.jsonc`:
|
|
39
44
|
|
|
40
45
|
```jsonc
|
|
41
46
|
{
|
|
42
|
-
"
|
|
43
|
-
"
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
"openai-codex/gpt-5.6-sol"
|
|
50
|
-
]
|
|
51
|
-
}
|
|
52
|
-
}
|
|
47
|
+
"gpt": {
|
|
48
|
+
"description": "Models available for this project",
|
|
49
|
+
"models": [
|
|
50
|
+
"openai-codex/gpt-5.6-luna",
|
|
51
|
+
"openai-codex/gpt-5.6-terra",
|
|
52
|
+
"openai-codex/gpt-5.6-sol"
|
|
53
|
+
]
|
|
53
54
|
}
|
|
54
55
|
}
|
|
55
56
|
```
|
|
56
57
|
|
|
57
58
|
The example agent selects Terra, not Luna: the agent's order wins. Sol is
|
|
58
59
|
available in the pool but absent from this worker's chain, so it cannot become
|
|
59
|
-
an automatic implementation fallback. The oracle
|
|
60
|
-
|
|
60
|
+
an automatic implementation fallback. The oracle and frontier-review can
|
|
61
|
+
declare Sol in their own chains. Model references in `models` must be exact
|
|
62
|
+
`provider/model` values, not wildcards.
|
|
61
63
|
|
|
62
64
|
The resolver intersects the agent chain with the selected pool. Runtime
|
|
63
65
|
selection then skips unregistered, unauthenticated or session-exhausted models.
|
|
64
|
-
Tasks with images and
|
|
66
|
+
Tasks with images and UI QA require confirmed image support. The first
|
|
65
67
|
eligible candidate runs; only the remaining eligible candidates are passed to
|
|
66
68
|
quota fallback. An empty intersection or unavailable chain rejects the batch
|
|
67
69
|
before any children or run state are created. Model selection makes no LLM
|
|
@@ -83,22 +85,37 @@ Use `/subagent-preset <name>`, `AGENTS_PRESET=<name>` or
|
|
|
83
85
|
without a pool filter. The shipped names remain compatible with saved choices:
|
|
84
86
|
`cheap` is the GLM pool, `gpt` the GPT pool, and `deep` the mixed pool. The last
|
|
85
87
|
name no longer means that ordinary workers should escalate to flagship models.
|
|
88
|
+
Bundled definitions live beside the built-in agents in
|
|
89
|
+
`src/async-subagents/agents/presets.jsonc`; a project file with the same preset
|
|
90
|
+
name overrides that pool.
|
|
86
91
|
|
|
87
92
|
Old role names are not implicit aliases. `quick`, `scan`, `review`, `deep`,
|
|
88
93
|
`docs`, `frontend`, and `tests` work only when explicitly defined as ordinary
|
|
89
94
|
custom/project types. This keeps the effective catalog and accepted names exact.
|
|
90
95
|
|
|
91
|
-
|
|
92
|
-
`
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
96
|
+
Agent frontmatter can gate whether a role exists for the current parent model:
|
|
97
|
+
`forParentModels` is an optional allow-list and `notForParentModels` is an
|
|
98
|
+
optional deny-list; deny wins when both match. These fields accept model
|
|
99
|
+
patterns such as `zai/*` and affect the parent catalog, explicit role
|
|
100
|
+
validation, and automatic routing. They do not change which model the child
|
|
101
|
+
runs on; `models` / legacy model selectors still own child model selection.
|
|
102
|
+
|
|
103
|
+
Legacy `model` plus `fallbackModels` and `modelByParent` still load when they are
|
|
104
|
+
declared in an agent Markdown file. New profile `models` replaces inherited
|
|
105
|
+
legacy selection fields. Empty `models` means no candidates, not permission to
|
|
106
|
+
inherit the parent model. Model-less project specialists must declare candidates
|
|
107
|
+
or receive an explicit model override.
|
|
108
|
+
|
|
109
|
+
Legacy singular selectors are normalized with an explicit fallback array:
|
|
110
|
+
`model` without `fallbackModels` resolves to `fallbackModels: []`, and every
|
|
111
|
+
normalized `modelByParent` entry carries its own `fallbackModels` array. Modern
|
|
112
|
+
`models` profiles already encode the complete ordered candidate/fallback chain
|
|
113
|
+
in one array and are not wrapped in an additional fallback field.
|
|
114
|
+
|
|
115
|
+
The removed `asyncSubagents` section is no longer part of the public config
|
|
116
|
+
schema and is not read at runtime. Existing legacy files are left untouched but
|
|
117
|
+
have no effect. Migrate role definitions to `<project>/.pi/agents/*.md` and
|
|
118
|
+
custom pools to `<project>/.pi/agents/presets.jsonc`.
|
|
102
119
|
|
|
103
120
|
## Compact handoff
|
|
104
121
|
|
|
@@ -106,4 +123,5 @@ Give workers a scope, acceptance criteria and the evidence needed to start.
|
|
|
106
123
|
Read compact results first and inspect raw artifacts selectively. One noisy
|
|
107
124
|
sequential investigation can justify a worker; a command whose exit status is
|
|
108
125
|
sufficient usually only needs a saved log, not another LLM. Independent review
|
|
109
|
-
|
|
126
|
+
of substantive code changes uses `frontier-review` when it is present in the
|
|
127
|
+
current parent catalog; use `research` for focused evidence/review questions.
|
|
@@ -0,0 +1,440 @@
|
|
|
1
|
+
# UI QA sub-agent specification
|
|
2
|
+
|
|
3
|
+
## Type
|
|
4
|
+
|
|
5
|
+
As-is
|
|
6
|
+
|
|
7
|
+
## Lifecycle
|
|
8
|
+
|
|
9
|
+
Active implemented contract.
|
|
10
|
+
|
|
11
|
+
## Goal
|
|
12
|
+
|
|
13
|
+
Provide a cheap, fast `ui-qa` async-subagent that reproduces user-visible bugs
|
|
14
|
+
and proves fixes across browser/web UI, terminal/TUI applications, and native
|
|
15
|
+
desktop GUIs. Browser QA selects between the existing trusted Playwright runner
|
|
16
|
+
for ordinary isolated E2E/auth/video/trace work and a capability-probed Chrome
|
|
17
|
+
DevTools CLI provider for DevTools-specific AX, console, network, performance,
|
|
18
|
+
Lighthouse, and memory diagnostics. Native/TUI QA uses a real PTY or
|
|
19
|
+
deterministic app/platform UI driver and retains inspectable captures plus
|
|
20
|
+
automatic bounded video evidence when the backend supports it.
|
|
21
|
+
Its ranked `models` list prefers `zai/glm-5.3-flash`, then
|
|
22
|
+
`openai-codex/gpt-5.6-luna`, filtered by the active preset's model pool and
|
|
23
|
+
confirmed runtime image support.
|
|
24
|
+
|
|
25
|
+
## Inline agent workflow and skill isolation
|
|
26
|
+
|
|
27
|
+
- The bundled role is a thin common contract: `src/async-subagents/agents/ui-qa.md`
|
|
28
|
+
keeps only the shared invariants (real target, backend classification,
|
|
29
|
+
`BLOCKED` semantics, deterministic-assertion oracle, bounded execution, owned
|
|
30
|
+
cleanup, private evidence) plus the base-guide/probe/run invocation syntax.
|
|
31
|
+
It deliberately omits browser-provider, terminal-presentation,
|
|
32
|
+
desktop-platform, and credential implementation details. Its body becomes the
|
|
33
|
+
QA child's `promptAppend` through the shared agent loader. Parent and router
|
|
34
|
+
catalogs include only its short `description`.
|
|
35
|
+
- Backend-specific instructions live in the canonical resource tree under
|
|
36
|
+
`src/async-subagents/agents/ui-qa/guides/`. `browser.md`, `tui.md`, and
|
|
37
|
+
`desktop.md` are compact base routers. Detail topics are browser
|
|
38
|
+
`playwright`/`chrome-devtools`/`auth`, TUI `pty`/`native-terminal`, and desktop
|
|
39
|
+
`macos-accessibility`/`windows-uia`/`linux-at-spi`. The child first loads only
|
|
40
|
+
its matching base guide through
|
|
41
|
+
`node "$PI_UI_QA_RUNNER" guide --backend browser|tui|desktop`, then only the
|
|
42
|
+
routed detail topic. The command resolves bundled files from a fixed
|
|
43
|
+
backend-scoped allowlist, rejects unknown or cross-backend topics, unknown
|
|
44
|
+
options, extra arguments, and traversal, bounds guide size, and prints only
|
|
45
|
+
the requested document; the model never composes or reads source paths.
|
|
46
|
+
- The capability-first runner lives under
|
|
47
|
+
`src/async-subagents/agents/ui-qa/`, with browser, PTY/TUI, and platform
|
|
48
|
+
desktop accessibility backends for macOS, Windows, and Linux. The trusted
|
|
49
|
+
Playwright browser runner, vendor dependencies/licenses, and optional legacy
|
|
50
|
+
JSONC examples are colocated under `agents/ui-qa/browser/`. The Chrome
|
|
51
|
+
DevTools provider is a bundled adapter under `agents/ui-qa/drivers/` and
|
|
52
|
+
capability-probes an external `chrome-devtools` CLI rather than loading a
|
|
53
|
+
project skill. None of these assets, including the guides, is a discoverable
|
|
54
|
+
skill or agent role (`agents/*.md` discovery stays non-recursive and
|
|
55
|
+
top-level only).
|
|
56
|
+
- Sub-agent processes disable normal extension discovery, then always load the
|
|
57
|
+
suite's model-tools extension. They load the Antigravity provider extension
|
|
58
|
+
only when an Antigravity model is explicitly selected. The launcher appends
|
|
59
|
+
`--models <effective-model>` after forwarded arguments so persisted model
|
|
60
|
+
patterns cannot resolve unrelated providers inside the isolated child.
|
|
61
|
+
- Every async sub-agent launches with `--no-skills`. `--skill` and
|
|
62
|
+
`--skill=...` flags are removed from `extraArgs`, and role profiles have no
|
|
63
|
+
skill-loading field. The thin agent Markdown plus its on-demand bundled
|
|
64
|
+
guides are the complete role instruction source.
|
|
65
|
+
- Explicit legacy `browser-qa` tasks normalize to `ui-qa`; a project-local
|
|
66
|
+
`browser-qa.md` profile override is migrated onto the canonical `ui-qa`
|
|
67
|
+
profile during config loading. Installed browser resources are part of the
|
|
68
|
+
canonical `ui-qa` tree, while the private runtime workspace keeps its
|
|
69
|
+
historical `browser-qa/` name for compatibility.
|
|
70
|
+
- The launcher sets `PI_UI_QA_RUNNER` to the absolute capability-first runner
|
|
71
|
+
and `PI_BROWSER_QA_RUNNER` to its trusted browser backend, replacing inherited
|
|
72
|
+
values and stripping both from ordinary children. QA uses the unified runner
|
|
73
|
+
for backend probe/run and the browser runner directly only for credential
|
|
74
|
+
profile discovery or form-auth scaffolding.
|
|
75
|
+
- Model-only profile overrides inherit the workflow. An explicit profile
|
|
76
|
+
`promptAppend` replaces the body like any other agent profile; it is not an
|
|
77
|
+
immutable security boundary. Runtime protections remain in the runner, and
|
|
78
|
+
the thin prompt's guide-routing requirement does not weaken them: every
|
|
79
|
+
backend action still goes through the runner's fail-closed checks.
|
|
80
|
+
|
|
81
|
+
## Browser authentication contract
|
|
82
|
+
|
|
83
|
+
- Public browser QA requires no auth profile and does not create or require
|
|
84
|
+
`.pi/qa_auth.jsonc`. Its explicit base URL supplies the one exact allowed
|
|
85
|
+
origin, and the runner still blocks every other HTTP(S)/WebSocket origin.
|
|
86
|
+
- Auth profiles live in project-local `.pi/qa_auth.jsonc` and are selected by
|
|
87
|
+
explicit id. The file must be a real project-local file with mode `0600` on
|
|
88
|
+
POSIX. Profile listings expose only `id`, description, and traits.
|
|
89
|
+
- Listing profiles when the file is absent returns an empty list without side
|
|
90
|
+
effects. When authenticated QA explicitly requests credentials and that file
|
|
91
|
+
is absent, the runner creates a private empty template and returns
|
|
92
|
+
`provide_credentials`.
|
|
93
|
+
The sub-agent must explicitly ask the user to fill the reported file and
|
|
94
|
+
rerun QA; it must not read or edit the credential values itself.
|
|
95
|
+
- Every profile requires one or more exact `allowedOrigins`. Secret-bearing auth
|
|
96
|
+
is applied only to those origins; all other HTTP(S)/WebSocket traffic and
|
|
97
|
+
service workers are blocked during QA.
|
|
98
|
+
- Supported auth types are `form`, `cookie`, `localStorage`, `sessionStorage`,
|
|
99
|
+
`bearer`, and existing Playwright `storageState`.
|
|
100
|
+
- The bundled runner reads secrets internally. Credentials must never be copied
|
|
101
|
+
into prompts, generated QA flows, shell arguments, transcripts, reports,
|
|
102
|
+
or QA evidence.
|
|
103
|
+
- Generated browser state is private cache under `.pi/qa-auth-state`. Ephemeral
|
|
104
|
+
flows, evidence, and result manifests are written under the owning agent's
|
|
105
|
+
`.pi/subagents/<run>/<agent-id>/browser-qa/` workspace. Multiple profiles use
|
|
106
|
+
separate browser contexts/evidence directories, and normal sub-agent shutdown
|
|
107
|
+
or cleanup removes the whole workspace with its run.
|
|
108
|
+
- Missing, rejected, or expired explicitly selected auth returns a
|
|
109
|
+
machine-readable update-required status naming only the profile id, config
|
|
110
|
+
file, and redacted reason. The parent asks the user to update the file and
|
|
111
|
+
reruns; there is no `/qa-auth` command.
|
|
112
|
+
|
|
113
|
+
## Native/TUI execution contract
|
|
114
|
+
|
|
115
|
+
- The QA target must be the actual user-facing TUI or desktop application named
|
|
116
|
+
by the task. Repository tests, snapshots, source inspection, or a different
|
|
117
|
+
CLI/web surface may support discovery but cannot substitute for requested UI
|
|
118
|
+
execution.
|
|
119
|
+
- The launcher creates a private `ui-qa/` workspace under the owning agent
|
|
120
|
+
directory. Native/TUI transcripts, captures, screenshots, videos, and small
|
|
121
|
+
temporary driver artifacts stay there and are removed with the sub-agent run.
|
|
122
|
+
- Repeated unnamed Desktop evidence steps receive collision-free filenames
|
|
123
|
+
based on their action and step index. Explicit names remain available when a
|
|
124
|
+
stable human-readable artifact label is useful.
|
|
125
|
+
- Terminal/TUI verification is selected from `target.command`, with an explicit
|
|
126
|
+
presentation contract that is independent of project/app identity.
|
|
127
|
+
New QA flows choose presentation explicitly by surface category: `pty` is for
|
|
128
|
+
line-oriented/plain terminal/CLI programs or protocol-focused semantic tests,
|
|
129
|
+
while `native-terminal` is the normal presentation for structured/full-screen
|
|
130
|
+
TUIs. Omitted presentation remains a `pty` compatibility default for older
|
|
131
|
+
flows only. Native-terminal covers real-window colors, fonts/glyphs, special
|
|
132
|
+
symbols, wrapping, clipping, menus/focus, and pixel geometry.
|
|
133
|
+
The target still runs in the runner-owned PTY used for deterministic input and
|
|
134
|
+
semantic assertions; the runner mirrors that same raw PTY byte stream through
|
|
135
|
+
a private authenticated local bridge into a fresh owned native terminal host,
|
|
136
|
+
whose real window supplies screenshots/video. The bridge bootstrap never
|
|
137
|
+
contains the target argv/cwd/env. Native-terminal provider discovery is
|
|
138
|
+
environment/capability based, not project based: macOS prefers installed
|
|
139
|
+
iTerm2 then Terminal.app; Windows uses Windows Terminal; Linux prefers kitty,
|
|
140
|
+
then Alacritty, GNOME Terminal, Konsole, and xterm. The provider is usable for
|
|
141
|
+
visual QA only when the corresponding desktop driver can also correlate and
|
|
142
|
+
capture its real window.
|
|
143
|
+
Selection is based on required capabilities/evidence, never a repository-
|
|
144
|
+
specific heuristic. If pixel fidelity is required but native-terminal control
|
|
145
|
+
is unavailable, the result is `BLOCKED`; the runner does not silently
|
|
146
|
+
substitute a headless replay.
|
|
147
|
+
- The target in either TUI presentation may use its normal explicit
|
|
148
|
+
project/session arguments to open deterministic state before assertions.
|
|
149
|
+
Non-interactive stdout from another CLI path is not TUI verification.
|
|
150
|
+
Automated input in both presentations goes to the same owned PTY, and native-
|
|
151
|
+
terminal host stdin/protocol responses are bridged back to it; flows wait for
|
|
152
|
+
expected text or a stable frame after `sendText`/`sendKeys` before asserting
|
|
153
|
+
or capturing the resulting state.
|
|
154
|
+
- Native desktop verification is selected from `target.application`. macOS uses
|
|
155
|
+
the bundled Accessibility/CGWindow/ScreenCaptureKit helper, Windows uses the
|
|
156
|
+
bundled PowerShell/.NET UI Automation helper, and Linux uses the bundled
|
|
157
|
+
Python AT-SPI helper when its runtime dependencies and graphical accessibility
|
|
158
|
+
bus are available. Required runtime components are capability-probed; missing
|
|
159
|
+
dependencies/permissions or unsupported platforms return `BLOCKED`. The agent
|
|
160
|
+
must not install UI automation dependencies, change OS privacy/accessibility
|
|
161
|
+
permissions, disable sandboxing, or operate unrelated user windows.
|
|
162
|
+
- PTY-presentation runs automatically retain a bounded asciicast v2 replay in
|
|
163
|
+
`artifacts.videos`. It is generated from timestamped PTY output and resize
|
|
164
|
+
events, capped at 1 MiB, and is explicitly terminal-state replay rather than
|
|
165
|
+
pixel evidence. Native-terminal presentation instead retains real-window
|
|
166
|
+
screenshots and, when exact-window capture is available, a bounded MP4 from
|
|
167
|
+
the owned native terminal host; its asciicast/headless captures remain
|
|
168
|
+
diagnostic-only and are never promoted as proof of colors/glyphs/window
|
|
169
|
+
geometry.
|
|
170
|
+
- When macOS 12.3+ ScreenCaptureKit and Screen Recording permission are
|
|
171
|
+
available, desktop runs automatically retain a silent H.264 MP4 of only the
|
|
172
|
+
correlated application window. Independent-window capture scales to fill the
|
|
173
|
+
Retina encoder surface so the application occupies the complete video frame
|
|
174
|
+
rather than a top-left subset with unused canvas. Recording is capped at 30
|
|
175
|
+
seconds, has no display/region fallback, and is best-effort: an unavailable
|
|
176
|
+
video is reported as a structured observation rather than an assertion
|
|
177
|
+
failure.
|
|
178
|
+
- Windows UIA provides exact-window PNG capture through trusted Win32 APIs but
|
|
179
|
+
does not yet advertise exact-window video. Linux AT-SPI provides PNG capture
|
|
180
|
+
only when a supported screenshot producer (`gnome-screenshot` or `scrot`) is
|
|
181
|
+
available; keyboard input is separately gated on a working `xdotool` in the
|
|
182
|
+
current graphical session. Linux/Windows video remains an explicit missing
|
|
183
|
+
capability rather than being synthesized from a different surface.
|
|
184
|
+
- Pass/fail requires a deterministic product-visible oracle such as terminal
|
|
185
|
+
content/state, accessibility/app-driver control state, window/dialog state,
|
|
186
|
+
visible copy, enabled/checked/value state, or another explicit application
|
|
187
|
+
result. Screenshots explain the result but are not the sole oracle.
|
|
188
|
+
- When no safe deterministic PTY/GUI control path is available, the correct
|
|
189
|
+
result is `BLOCKED`; static tests are not promoted to UI QA evidence.
|
|
190
|
+
- Cleanup is ownership-scoped: terminate only the PTY/session/app/driver process
|
|
191
|
+
created by the run, never all processes with a matching application name.
|
|
192
|
+
POSIX desktop launch contracts correlate and clean up the complete detached
|
|
193
|
+
process group, so a package-manager wrapper may hand off to its GUI descendant
|
|
194
|
+
without making that app unreachable or leaving it running. Windows uses the
|
|
195
|
+
owned launcher PID as a process-tree root: UIA resolves the actual GUI
|
|
196
|
+
descendant before interaction, and scoped cleanup uses `taskkill /T` against
|
|
197
|
+
the owned launcher plus that correlated GUI root rather than an app name.
|
|
198
|
+
|
|
199
|
+
## Unified capability-first runner contract
|
|
200
|
+
|
|
201
|
+
- One private JSONC flow under the owning agent's `ui-qa/flows/` declares
|
|
202
|
+
exactly one browser URL, TUI command, or desktop application target. `probe`
|
|
203
|
+
reports deterministic candidate capabilities and selects the matching backend;
|
|
204
|
+
it also returns an authoritative `selection.guide = {backend, topic}` for the
|
|
205
|
+
selected provider/presentation/platform detail when one exists. The child
|
|
206
|
+
must load/reconcile that exact topic before `run`, which then executes the
|
|
207
|
+
bounded flow. The child keeps the flow at mode `0600` on POSIX before either
|
|
208
|
+
command, matching the runner's private-path checks.
|
|
209
|
+
- Browser execution adapts the unified target and steps to the existing trusted
|
|
210
|
+
Playwright runner for ordinary E2E or to the Chrome DevTools provider for
|
|
211
|
+
DevTools-only capabilities. `target.browserDriver` is `auto`, `playwright`,
|
|
212
|
+
or `chrome-devtools`; `auto` keeps ordinary flows and every trusted auth
|
|
213
|
+
profile on Playwright, and selects DevTools only when the flow requests a
|
|
214
|
+
DevTools-only action or explicit `target.devtools` attach/start options.
|
|
215
|
+
- TUI execution launches only a bounded project-local/package-runtime contract
|
|
216
|
+
through a real PTY, models ANSI/VT alternate-screen state with a headless
|
|
217
|
+
terminal, and supports text, cursor, process, resize, stability, and capture
|
|
218
|
+
assertions.
|
|
219
|
+
- Desktop execution is platform-specific behind one capability contract. macOS
|
|
220
|
+
uses a bundled compiled Accessibility/CGWindow helper plus ScreenCaptureKit;
|
|
221
|
+
Windows uses bundled PowerShell/.NET UI Automation plus Win32 window capture;
|
|
222
|
+
Linux uses bundled Python AT-SPI plus runtime-probed screenshot/keyboard
|
|
223
|
+
producers. Explicit selectors and owned launch roots must still resolve the
|
|
224
|
+
actual GUI descendant. Unsupported platforms or missing required control
|
|
225
|
+
capabilities return a structured `BLOCKED` result; unavailable best-effort
|
|
226
|
+
video is reported without replacing deterministic assertions.
|
|
227
|
+
- Results normalize selection rationale, assertions, observations, and typed
|
|
228
|
+
artifact groups across all backends. `BLOCKED` additionally normalizes a
|
|
229
|
+
parent-facing `blockedHandoff` containing the selected backend/platform
|
|
230
|
+
driver, missing capabilities, concrete reason, remediation string,
|
|
231
|
+
`manualActionRequired: true`, and
|
|
232
|
+
`automaticRemediationAttempted: false`. This handoff is the installation/
|
|
233
|
+
permission/platform-repair reference for the parent; the QA child relays it
|
|
234
|
+
and never performs those environment changes itself. Every runner/app/helper
|
|
235
|
+
process has a bounded deadline and cleanup is limited to processes launched by
|
|
236
|
+
that run.
|
|
237
|
+
|
|
238
|
+
## Browser backend execution contract
|
|
239
|
+
|
|
240
|
+
- A model-authored QA flow is declarative JSONC, not executable JavaScript. The
|
|
241
|
+
selected provider implements a bounded set of navigation, interaction,
|
|
242
|
+
assertion, and evidence actions. The flow never receives a Playwright context,
|
|
243
|
+
DevTools protocol object, arbitrary JavaScript evaluator, or credential
|
|
244
|
+
values.
|
|
245
|
+
- Browser provider selection is capability-based and project-agnostic. The
|
|
246
|
+
Playwright provider remains the default for ordinary repeatable E2E, trusted
|
|
247
|
+
authentication, downloads, frames/popups, rich locators, deterministic
|
|
248
|
+
locale/timezone/motion settings, automatic video, and sanitized Playwright
|
|
249
|
+
trace evidence. Chrome DevTools is selected for structured accessibility-tree
|
|
250
|
+
snapshots, console/network assertions, Lighthouse summaries, sanitized
|
|
251
|
+
summaries from Chrome performance traces, and sanitized heap summaries.
|
|
252
|
+
Explicitly forcing a
|
|
253
|
+
provider that cannot implement the requested action returns `BLOCKED` rather
|
|
254
|
+
than weakening the scenario.
|
|
255
|
+
- Chrome DevTools requires `chrome-devtools-mcp >= 1.9.0` on `PATH`. Probe only
|
|
256
|
+
checks the CLI/version; run creates a random per-run daemon `sessionId` so its
|
|
257
|
+
socket/PID lifecycle does not collide with a user's existing DevTools daemon.
|
|
258
|
+
The runner disables JavaScript evaluation, extension/PWA/experimental tool
|
|
259
|
+
categories, usage statistics, and CrUX lookups; enables network-header
|
|
260
|
+
redaction and page-id routing; restricts filesystem writes to the run evidence
|
|
261
|
+
directory; and passes every exact allowed origin to the DevTools network
|
|
262
|
+
allowlist. Raw network headers/response bodies are not retained or surfaced.
|
|
263
|
+
- Chrome DevTools starts an isolated browser/profile by default. Optional
|
|
264
|
+
`target.devtools.browserUrl` accepts only a credential-free local loopback
|
|
265
|
+
HTTP origin. Attached Chrome still gets a task-owned isolated page/context by
|
|
266
|
+
default. `reuseExistingBrowserSession: true` is allowed only with that
|
|
267
|
+
loopback endpoint and is reserved for an explicitly requested reuse of the
|
|
268
|
+
user's already-authenticated Chrome session. Cleanup closes the task-created
|
|
269
|
+
page and stops only the private daemon; it never stops the external Chrome or
|
|
270
|
+
closes unrelated tabs.
|
|
271
|
+
- Trusted `.pi/qa_auth.jsonc` profiles remain Playwright-only. Chrome DevTools
|
|
272
|
+
cannot consume a QA auth profile, and credentials may not be moved into a
|
|
273
|
+
DevTools browser profile, command arguments, environment, or generated flow.
|
|
274
|
+
A missing/old CLI produces the standard parent-ready `blockedHandoff`; the QA
|
|
275
|
+
child does not install or upgrade it.
|
|
276
|
+
- Chrome DevTools supports the common `goto`, `reload`, `click`, `doubleClick`,
|
|
277
|
+
`hover`, `fill`, `press`, `waitFor`, `waitForTimeout`, `assertVisible`,
|
|
278
|
+
`assertText`, `assertURL`, and `screenshot` subset plus
|
|
279
|
+
`snapshotAccessibility`, `assertNoConsoleErrors`, `assertConsole`,
|
|
280
|
+
`assertNetworkRequest`, `lighthouse`, `performanceTrace`, and `heapSummary`.
|
|
281
|
+
Its locators are intentionally limited to AX `{role,name?,exact?}` or
|
|
282
|
+
`{text,exact?}` and must resolve to one UID from the latest snapshot. Raw heap
|
|
283
|
+
snapshots and raw performance traces are temporary and deleted after bounded
|
|
284
|
+
sanitized summaries are retained. Top-level `environment` remains a
|
|
285
|
+
Playwright-only deterministic
|
|
286
|
+
contract; DevTools blocks it rather than silently approximating
|
|
287
|
+
locale/timezone/reduced-motion behavior.
|
|
288
|
+
- Target discovery is a bounded preflight, not an open-ended research task. The
|
|
289
|
+
sub-agent invokes the runner within 45 seconds or returns `BLOCKED`; it does
|
|
290
|
+
not spend the full launcher budget reading source or probing prerequisites.
|
|
291
|
+
- The launcher injects `PI_SUBAGENT_AGENT_DIR`, pre-creates private `ui-qa/` and
|
|
292
|
+
`browser-qa/flows/` workspaces, and clears stale UI/browser QA files when an
|
|
293
|
+
agent id is reused. The runner validates the directory's project/type
|
|
294
|
+
metadata and refuses flows outside it; the model cannot select a shared
|
|
295
|
+
evidence root.
|
|
296
|
+
- The runner owns browser lifecycle, origin checks, auth application, tracing,
|
|
297
|
+
screenshots, video finalization, and redacted result output. Before retaining
|
|
298
|
+
a trace it removes network/non-image resource entries, redacts configured and
|
|
299
|
+
runtime storage credentials, and verifies those values are absent.
|
|
300
|
+
- After every navigation or visible interaction, the runner waits for DOM
|
|
301
|
+
readiness, completion of requests started by the action, and disappearance of
|
|
302
|
+
common visible busy/spinner/skeleton markers. It requires a 500 ms stable
|
|
303
|
+
interval before the next action so recordings remain readable; a page that
|
|
304
|
+
stays busy through the flow timeout fails closed instead of being tested as a
|
|
305
|
+
loading shell. App-specific readiness still requires an explicit declarative
|
|
306
|
+
wait/assertion in the authored flow.
|
|
307
|
+
- Before any page is created, the runner installs a context-wide, isolated
|
|
308
|
+
interaction visualizer. Recorded clicks/double-clicks show a transient cursor
|
|
309
|
+
and pulse. Native drag/drop is replayed for 450 ms with a large orange cursor,
|
|
310
|
+
progressively drawn high-contrast path, and green drop marker. It also covers
|
|
311
|
+
same-origin frames, declared popups, and form-auth submission. The layer is
|
|
312
|
+
accessibility-hidden, pointer-transparent, never cancels application events,
|
|
313
|
+
and its bounded animations clear within the post-action stable interval.
|
|
314
|
+
- Success and post-launch failure results include typed artifact groups. Every
|
|
315
|
+
item has an absolute filesystem path and a `file:` URI; the sub-agent must
|
|
316
|
+
present each item as a clickable Markdown link instead of reporting only the
|
|
317
|
+
evidence directory.
|
|
318
|
+
- Success requires deterministic assertions. Visual inspection supplements,
|
|
319
|
+
but never replaces, explicit expected-state checks.
|
|
320
|
+
- Auth rejection discovered by a QA flow is reported through the
|
|
321
|
+
`authRejectedIf` action so the parent gets an update-required status.
|
|
322
|
+
|
|
323
|
+
## Reliability and shutdown contract
|
|
324
|
+
|
|
325
|
+
- The built-in `ui-qa` profile has a 300-second wall-clock budget unless
|
|
326
|
+
the caller explicitly supplies a task or spawn timeout. This bounds model
|
|
327
|
+
stalls as well as browser work.
|
|
328
|
+
- The trusted runner has its own bounded lifecycle. Browser launch, context
|
|
329
|
+
setup, auth, flow execution, evidence finalization, and browser shutdown must
|
|
330
|
+
not wait forever; a timeout reports the last started stage without exposing
|
|
331
|
+
flow contents or credentials.
|
|
332
|
+
- Trace sanitization runs in a memory-limited worker that can be terminated at
|
|
333
|
+
the cleanup deadline; synchronous archive work cannot defeat the watchdog.
|
|
334
|
+
- The launcher always writes a small sanitized `progress.jsonl` journal in the
|
|
335
|
+
agent directory. It records lifecycle/RPC event types and tool names, but not
|
|
336
|
+
prompts, tool arguments, tool results, model text, or secrets. The browser
|
|
337
|
+
runner writes similarly sanitized stage entries under its private workspace.
|
|
338
|
+
- On POSIX, newly launched agents own a process group. Settled, timed-out, and
|
|
339
|
+
explicitly stopped agents signal that group rather than only the Pi process;
|
|
340
|
+
timeout/settled shutdown escalates to `SIGKILL` after its grace period. On
|
|
341
|
+
Windows the existing recursive `taskkill /T /F` behavior remains in force.
|
|
342
|
+
- Process-tree cleanup is scoped to a launcher-created process-group marker so
|
|
343
|
+
an old or externally-created PID is never treated as an owned process group.
|
|
344
|
+
User browser sessions outside that group must not be signalled.
|
|
345
|
+
- Playwright can launch Chromium in its own POSIX process group. On runner
|
|
346
|
+
failure the runner snapshots and kills only its own descendants before it
|
|
347
|
+
exits, covering that detached browser tree without touching a user's browser.
|
|
348
|
+
|
|
349
|
+
## Related files
|
|
350
|
+
|
|
351
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa.md`
|
|
352
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/guides/`
|
|
353
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/scripts/ui-qa-runner.mjs`
|
|
354
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/backends/`
|
|
355
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/chrome-devtools/chrome-devtools-provider.mjs`
|
|
356
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/native-terminal/native-terminal-host.mjs`
|
|
357
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/native-terminal/bridge-client.mjs`
|
|
358
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/macos/macos-accessibility.swift`
|
|
359
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/windows/windows-uia.ps1`
|
|
360
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/drivers/linux/linux-atspi.py`
|
|
361
|
+
- `external/pi-tools-suite/src/async-subagents/core/browser-qa.ts`
|
|
362
|
+
- `external/pi-tools-suite/src/async-subagents/core/spawn.ts`
|
|
363
|
+
- `external/pi-tools-suite/src/async-subagents/agents/ui-qa/browser/scripts/browser-qa-runner.mjs`
|
|
364
|
+
- `external/pi-tools-suite/test/async-subagents/core.test.ts`
|
|
365
|
+
- `external/pi-tools-suite/test/async-subagents/browser-qa-runner.test.ts`
|
|
366
|
+
- `external/pi-tools-suite/test/async-subagents/browser-qa-runner.e2e.test.ts`
|
|
367
|
+
- `external/pi-tools-suite/test/async-subagents/ui-qa-runner.test.ts`
|
|
368
|
+
- `external/pi-tools-suite/test/async-subagents/ui-qa-desktop.e2e.test.ts`
|
|
369
|
+
- `external/pi-tools-suite/test/async-subagents/selection-e2e.test.ts`
|
|
370
|
+
|
|
371
|
+
## Acceptance criteria
|
|
372
|
+
|
|
373
|
+
1. `ui-qa` resolves to the intended model/fallback and its inline Markdown
|
|
374
|
+
workflow; explicit legacy `browser-qa` requests resolve to it, and its
|
|
375
|
+
isolated child process can register the configured
|
|
376
|
+
model provider.
|
|
377
|
+
2. Every child spawn contains `--no-skills` but no `--skill`. The QA child
|
|
378
|
+
receives the thin common contract plus guide-routing workflow in its initial
|
|
379
|
+
prompt, loads exactly one base backend guide and only its routed detail topic
|
|
380
|
+
through `PI_UI_QA_RUNNER guide`, treats `selection.guide` as authoritative
|
|
381
|
+
before `run`, and can reach the credential-owning browser backend through
|
|
382
|
+
`PI_BROWSER_QA_RUNNER` when the explicit auth topic requires it, from an
|
|
383
|
+
unrelated project directory.
|
|
384
|
+
Ordinary profiles are equally skill-free and do not receive QA-only
|
|
385
|
+
environment paths.
|
|
386
|
+
3. Auth profile listing and all error output are redacted; model-authored input
|
|
387
|
+
cannot execute code in the credential-bearing process.
|
|
388
|
+
4. Runner tests cover public execution without an auth file, explicit profile
|
|
389
|
+
selection, all auth modes, fail-closed origins, path/mode hardening, private
|
|
390
|
+
empty-template creation only on an explicit auth request, non-executable
|
|
391
|
+
flows, and successful redacted evidence creation.
|
|
392
|
+
5. Native/TUI evidence, including bounded PTY replay, native-terminal real-
|
|
393
|
+
window screenshots/video when requested and available, and exact-window
|
|
394
|
+
desktop video when available, lives under the owning agent's `ui-qa/`
|
|
395
|
+
workspace; browser flows/evidence remain under its browser-backend
|
|
396
|
+
`browser-qa/` workspace. Deleting the run removes both while persistent auth
|
|
397
|
+
config/state remains.
|
|
398
|
+
6. Runner tests prove that network activity and visible loading indicators are
|
|
399
|
+
awaited, persistent loading fails the flow, visible actions retain a stable
|
|
400
|
+
500 ms video interval, and context-wide click/drag video visualization is
|
|
401
|
+
installed with bounded click pacing.
|
|
402
|
+
7. Completed test runs report clickable screenshot, video, and trace links
|
|
403
|
+
whenever those artifacts exist.
|
|
404
|
+
8. Timeout tests identify the last browser stage, launcher progress remains
|
|
405
|
+
available when full RPC logging is disabled, and process-tree tests prove a
|
|
406
|
+
descendant is terminated without signalling unrelated processes.
|
|
407
|
+
9. TUI/native instructions require a real PTY/app driver, deterministic
|
|
408
|
+
product-visible oracles, scoped cleanup, and a `BLOCKED` result when safe
|
|
409
|
+
automation is unavailable rather than substituting source/unit tests.
|
|
410
|
+
Pixel-sensitive TUI tasks select native-terminal presentation by capability
|
|
411
|
+
need rather than app identity; the target still runs in one owned PTY while
|
|
412
|
+
a private bridge mirrors its bytes into an owned real terminal window.
|
|
413
|
+
10. Suite tests/typecheck, host checks, and suite sync pass.
|
|
414
|
+
11. Unified runner tests cover deterministic backend/presentation selection,
|
|
415
|
+
real PTY screen state and scoped cleanup, native-terminal bridge bootstrap
|
|
416
|
+
isolation, Windows/Linux native-host launch contracts, unsafe launch/path
|
|
417
|
+
rejection, timeout bounds, platform blockers, and the trusted Windows UIA/
|
|
418
|
+
Linux AT-SPI helper protocols. The opt-in macOS E2E launches a real AppKit
|
|
419
|
+
window, semantically activates its control, and retains accessibility,
|
|
420
|
+
screenshot, and automatic exact-window video evidence. Windows/Linux
|
|
421
|
+
real-host smokes are required before claiming those platform integrations
|
|
422
|
+
runtime-verified; macOS tests do not substitute for that evidence.
|
|
423
|
+
|
|
424
|
+
## Real-browser regression test
|
|
425
|
+
|
|
426
|
+
The repository includes a local mock-page E2E that launches real Chromium and
|
|
427
|
+
asserts PNG screenshots, WebM video, sanitized trace output, and absolute
|
|
428
|
+
path/`file:` URI metadata:
|
|
429
|
+
|
|
430
|
+
```bash
|
|
431
|
+
npx playwright install chromium
|
|
432
|
+
npm run test:browser-qa-e2e
|
|
433
|
+
```
|
|
434
|
+
|
|
435
|
+
Normal suite tests keep this case skipped; the Publish workflow runs it on
|
|
436
|
+
Linux after installing Chromium. The runner writes into a temporary simulated
|
|
437
|
+
sub-agent directory. For manual inspection only, explicit E2E runs copy the
|
|
438
|
+
latest artifacts to `.pi/qa-runs/browser-qa-e2e/latest/` and print clickable
|
|
439
|
+
links; this test-only published copy is not the runtime storage contract. Set
|
|
440
|
+
`BROWSER_QA_KEEP_EVIDENCE=0` to skip that copy.
|
|
@@ -22,12 +22,14 @@
|
|
|
22
22
|
"smoke:tools": "PI_OFFLINE=1 pi --no-session -p \"ping\"",
|
|
23
23
|
"smoke": "npm run smoke:explicit && npm run smoke:auto && npm run smoke:tools",
|
|
24
24
|
"test": "bun test test",
|
|
25
|
+
"test:dcp-session-sim": "bun test test/dcp-session-sim-e2e.test.ts",
|
|
26
|
+
"test:dcp-reminder-e2e": "DCP_REMINDER_E2E=1 bun test test/prompt-evals/dcp-reminder-e2e.test.ts",
|
|
25
27
|
"test:browser-qa-e2e": "BROWSER_QA_RUNNER_E2E=1 bun test test/async-subagents/browser-qa-runner.e2e.test.ts",
|
|
26
28
|
"test:async-subagents-e2e": "ASYNC_SUBAGENTS_E2E=1 ASYNC_SUBAGENTS_DEBUG_LOGS=1 ASYNC_SUBAGENTS_MODEL=zai/glm-5-turbo bun test --concurrent --max-concurrency=30 test/async-subagents",
|
|
27
29
|
"test:async-subagents-selection-e2e": "ASYNC_SUBAGENTS_SELECTION_E2E=1 ASYNC_SUBAGENTS_MODEL=zai/glm-5-turbo bun test --concurrent --max-concurrency=30 test/async-subagents/selection-e2e.test.ts",
|
|
28
30
|
"test:prompt-evals:tool-selection": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=10 test/tool-selection-e2e.test.ts",
|
|
29
31
|
"test:prompt-evals:async": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/async-subagents/selection-e2e.test.ts test/prompt-evals/async-routing-e2e.test.ts",
|
|
30
|
-
"test:prompt-evals:dcp": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/prompt-evals/dcp-summary-e2e.test.ts",
|
|
32
|
+
"test:prompt-evals:dcp": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/prompt-evals/dcp-summary-e2e.test.ts test/prompt-evals/dcp-reminder-e2e.test.ts",
|
|
31
33
|
"test:prompt-evals": "PROMPT_EVAL_E2E=1 bun test --concurrent --max-concurrency=5 test/tool-selection-e2e.test.ts test/async-subagents/selection-e2e.test.ts test/prompt-evals",
|
|
32
34
|
"test:evals:contracts": "bun test test/evals/extension-contracts.test.ts test/evals/harness.test.ts",
|
|
33
35
|
"test:evals:live": "PI_TOOLS_SUITE_EVALS_LIVE=1 bun test --concurrent --max-concurrency=4 test/evals/live-evals.test.ts",
|
|
@@ -46,6 +48,9 @@
|
|
|
46
48
|
"check": "npm run typecheck && npm test && npm run smoke"
|
|
47
49
|
},
|
|
48
50
|
"dependencies": {
|
|
51
|
+
"@cortexkit/antigravity-auth-core": "2.2.1",
|
|
52
|
+
"@lydell/node-pty": "1.1.0",
|
|
53
|
+
"@xterm/headless": "^6.0.0",
|
|
49
54
|
"jsonc-parser": "^3.3.1",
|
|
50
55
|
"vscode-jsonrpc": "^8.2.1",
|
|
51
56
|
"vscode-languageserver-protocol": "^3.17.5"
|
|
@@ -3,6 +3,7 @@ import { promises as fs } from "node:fs";
|
|
|
3
3
|
import { homedir } from "node:os";
|
|
4
4
|
import { basename, dirname, join } from "node:path";
|
|
5
5
|
import { getAgentDir } from "@earendil-works/pi-coding-agent";
|
|
6
|
+
import { ANTIGRAVITY_CLIENT_ID, ANTIGRAVITY_CLIENT_SECRET } from "@cortexkit/antigravity-auth-core";
|
|
6
7
|
import { DEFAULT_PROJECT_ID, PROVIDER_ID } from "./constants";
|
|
7
8
|
import type { GoogleOAuthClientCredentials, OpencodeAntigravityAccount, OpencodeAntigravityImportResult, OpencodeAntigravityStorage, PiAuthCredential, PiAuthData } from "./types";
|
|
8
9
|
|
|
@@ -113,7 +114,7 @@ export function getGoogleOAuthClientCredentials(...sources: Array<unknown>): Goo
|
|
|
113
114
|
const clientId = process.env.PI_ANTIGRAVITY_GOOGLE_CLIENT_ID;
|
|
114
115
|
const clientSecret = process.env.PI_ANTIGRAVITY_GOOGLE_CLIENT_SECRET;
|
|
115
116
|
if (clientId) return { clientId, ...(clientSecret ? { clientSecret } : {}) };
|
|
116
|
-
return
|
|
117
|
+
return { clientId: ANTIGRAVITY_CLIENT_ID, clientSecret: ANTIGRAVITY_CLIENT_SECRET };
|
|
117
118
|
}
|
|
118
119
|
|
|
119
120
|
export function clampAccountIndex(index: unknown, accountCount: number): number {
|
|
@@ -4,6 +4,10 @@ export const STATUS_KEY = "dcp:antigravity";
|
|
|
4
4
|
export const LEGACY_STATUS_KEY = "antigravity";
|
|
5
5
|
export const ALL_ACCOUNTS_EXHAUSTED_MARKER = "ANTIGRAVITY_ALL_ACCOUNTS_EXHAUSTED";
|
|
6
6
|
|
|
7
|
+
// Captured native agy CLI wire identity used by cortexkit 2.2.1.
|
|
8
|
+
export const AGY_CLI_VERSION = "1.1.24";
|
|
9
|
+
export const AGY_CLI_CHANGE_LIST = "974782877";
|
|
10
|
+
|
|
7
11
|
export const REDIRECT_URI = "http://localhost:51121/oauth-callback";
|
|
8
12
|
export const SCOPES = [
|
|
9
13
|
"https://www.googleapis.com/auth/cloud-platform",
|
|
@@ -13,11 +17,13 @@ export const SCOPES = [
|
|
|
13
17
|
"https://www.googleapis.com/auth/experimentsandconfigs",
|
|
14
18
|
];
|
|
15
19
|
|
|
16
|
-
export const ENDPOINT_DAILY = "https://daily-cloudcode-pa.
|
|
20
|
+
export const ENDPOINT_DAILY = "https://daily-cloudcode-pa.googleapis.com";
|
|
17
21
|
export const ENDPOINT_PROD = "https://cloudcode-pa.googleapis.com";
|
|
18
22
|
export const ENDPOINT_AUTOPUSH = "https://autopush-cloudcode-pa.sandbox.googleapis.com";
|
|
19
|
-
|
|
20
|
-
|
|
23
|
+
// Autopush is retained as a named legacy endpoint for compatibility/debugging,
|
|
24
|
+
// but current agy/cortexkit traffic falls back daily -> prod only.
|
|
25
|
+
export const STREAM_ENDPOINTS = [ENDPOINT_DAILY, ENDPOINT_PROD];
|
|
26
|
+
export const LOAD_ENDPOINTS = [ENDPOINT_DAILY, ENDPOINT_PROD];
|
|
21
27
|
export const DEFAULT_PROJECT_ID = "rising-fact-p41fc";
|
|
22
28
|
export const TOKEN_EXPIRY_SKEW_MS = 5 * 60 * 1000;
|
|
23
29
|
export const SKIP_THOUGHT_SIGNATURE = "skip_thought_signature_validator";
|