pi-ui-extend 1.0.32 → 1.0.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/README.md +21 -0
  2. package/dist/app/app.js +0 -1
  3. package/dist/app/commands/command-host.d.ts +1 -2
  4. package/dist/app/commands/command-model-actions.js +5 -5
  5. package/dist/app/constants.d.ts +0 -1
  6. package/dist/app/constants.js +0 -34
  7. package/dist/app/popup/menu-items-controller.d.ts +1 -3
  8. package/dist/app/popup/menu-items-controller.js +10 -32
  9. package/dist/app/rendering/popup-menu-renderer.js +1 -1
  10. package/dist/app/runtime.js +1 -3
  11. package/dist/app/screen/file-link-opener.js +15 -49
  12. package/external/pi-tools-suite/README.md +29 -7
  13. package/external/pi-tools-suite/docs/browser-qa-subagent.md +24 -6
  14. package/external/pi-tools-suite/docs/session-recovery.md +84 -0
  15. package/external/pi-tools-suite/src/async-subagents/async-subagents.sample.jsonc +13 -9
  16. package/external/pi-tools-suite/src/async-subagents/core/config.ts +2 -2
  17. package/external/pi-tools-suite/src/async-subagents/core/spawn.ts +23 -4
  18. package/external/pi-tools-suite/src/async-subagents/private-skills/browser-qa/SKILL.md +40 -4
  19. package/external/pi-tools-suite/src/async-subagents/private-skills/browser-qa/references/qa-design.md +19 -0
  20. package/external/pi-tools-suite/src/async-subagents/private-skills/browser-qa/scripts/browser-qa-runner.mjs +444 -13
  21. package/external/pi-tools-suite/src/coding-discipline/index.ts +12 -2
  22. package/external/pi-tools-suite/src/default-pi-tools-suite-config.ts +6 -6
  23. package/external/pi-tools-suite/src/index.ts +1 -0
  24. package/external/pi-tools-suite/src/session-recovery/index.ts +674 -0
  25. package/external/pi-tools-suite/src/tool-descriptions.ts +42 -0
  26. package/package.json +4 -2
  27. package/skills/skill-creator/scripts/__pycache__/__init__.cpython-314.pyc +0 -0
  28. package/skills/skill-creator/scripts/__pycache__/aggregate_benchmark.cpython-314.pyc +0 -0
  29. package/skills/skill-creator/scripts/__pycache__/generate_report.cpython-314.pyc +0 -0
  30. package/skills/skill-creator/scripts/__pycache__/improve_description.cpython-314.pyc +0 -0
  31. package/skills/skill-creator/scripts/__pycache__/package_skill.cpython-314.pyc +0 -0
  32. package/skills/skill-creator/scripts/__pycache__/run_eval.cpython-314.pyc +0 -0
  33. package/skills/skill-creator/scripts/__pycache__/run_loop.cpython-314.pyc +0 -0
  34. package/skills/skill-creator/scripts/__pycache__/utils.cpython-314.pyc +0 -0
@@ -5,7 +5,7 @@
5
5
  // Global async-subagents routing.
6
6
  // Rule of thumb:
7
7
  // - GLM: cheap/fast discovery, grep-style scanning, docs/test inventory.
8
- // - Gemini: frontend/UI/UX, visual/product polish, and multimodal-heavy analysis.
8
+ // - GLM-5.3 Flash: frontend/UI/UX, visual/product polish, and multimodal-heavy analysis.
9
9
  // - GPT/Claude: review/audit, deep reasoning, tests, and implementation/refactoring.
10
10
  // Override exact model IDs here if your pi provider uses different names.
11
11
  "defaultType": "quick",
@@ -34,7 +34,11 @@
34
34
  // screenshot inspection should use lookup; this only keeps guidance honest.
35
35
  // Matching is case-insensitive and supports `*` wildcards.
36
36
  "vision": {
37
- "blindModelPatterns": ["zai/glm*", "glm*", "*/glm*"]
37
+ "blindModelPatterns": [
38
+ "zai/glm-4.5*", "glm-4.5*", "*/glm-4.5*",
39
+ "zai/glm-5-turbo*", "glm-5-turbo*", "*/glm-5-turbo*",
40
+ "zai/glm-5.3", "glm-5.3", "*/glm-5.3"
41
+ ]
38
42
  },
39
43
 
40
44
  // Optional per-type prompt customization (leave commented until needed):
@@ -109,14 +113,14 @@
109
113
 
110
114
  "presets": {
111
115
  "cheap": {
112
- "description": "Use cheap GLM/Gemini Flash models for text/code roles.",
116
+ "description": "Use GLM models by role, including GLM-5.3 Flash for multimodal work.",
113
117
  "types": {
114
118
  "quick": { "model": "zai/glm-4.5-air", "thinking": "off" },
115
119
  "scan": { "model": "zai/glm-4.5-air", "thinking": "off" },
116
120
  "research": { "model": "zai/glm-5-turbo", "thinking": "low" },
117
121
  "docs": { "model": "zai/glm-4.5-air", "thinking": "low" },
118
- "frontend": { "model": "antigravity/gemini-3-flash-preview", "fallbackModels": ["zai/glm-5.3"], "thinking": "medium" },
119
- "browser-qa": { "model": "openai-codex/gpt-5.6-luna", "fallbackModels": ["antigravity/gemini-3-flash-preview", "zai/glm-5.3"], "thinking": "low" },
122
+ "frontend": { "model": "zai/glm-5.3-flash", "thinking": "medium" },
123
+ "browser-qa": { "model": "zai/glm-5.3-flash", "fallbackModels": ["openai-codex/gpt-5.6-luna"], "thinking": "low" },
120
124
  "tests": { "model": "zai/glm-5-turbo", "thinking": "medium" },
121
125
  "review": { "model": "zai/glm-5.3", "thinking": "high" },
122
126
  "implement": { "model": "zai/glm-5.3", "thinking": "high" },
@@ -132,7 +136,7 @@
132
136
  "research": { "model": "openai-codex/gpt-5.6-terra", "fallbackModels": ["zai/glm-5-turbo"], "thinking": "low" },
133
137
  "docs": { "model": "openai-codex/gpt-5.6-luna", "fallbackModels": ["zai/glm-4.5-air"], "thinking": "low" },
134
138
  "frontend": { "model": "openai-codex/gpt-5.6-terra", "fallbackModels": ["antigravity/gemini-3-flash-preview", "zai/glm-5.3"], "thinking": "medium" },
135
- "browser-qa": { "model": "openai-codex/gpt-5.6-luna", "fallbackModels": ["antigravity/gemini-3-flash-preview", "zai/glm-5.3"], "thinking": "low" },
139
+ "browser-qa": { "model": "zai/glm-5.3-flash", "fallbackModels": ["openai-codex/gpt-5.6-luna"], "thinking": "low" },
136
140
  "tests": { "model": "openai-codex/gpt-5.6-terra", "fallbackModels": ["zai/glm-5-turbo"], "thinking": "medium" },
137
141
  "review": { "model": "openai-codex/gpt-5.6-sol", "fallbackModels": ["zai/glm-5.3"], "thinking": "high" },
138
142
  "implement": { "model": "openai-codex/gpt-5.6-sol", "fallbackModels": ["zai/glm-5.3"], "thinking": "high" },
@@ -148,7 +152,7 @@
148
152
  "research": { "model": "antigravity/gemini-3.1-pro-preview", "fallbackModels": ["openai-codex/gpt-5.6-luna", "zai/glm-5-turbo"], "thinking": "medium" },
149
153
  "docs": { "model": "antigravity/gemini-2.5-flash", "fallbackModels": ["openai-codex/gpt-5.6-luna", "zai/glm-4.5-air"], "thinking": "medium" },
150
154
  "frontend": { "model": "antigravity/gemini-3.1-pro-preview-customtools", "fallbackModels": ["openai-codex/gpt-5.6-luna", "zai/glm-5.3"], "thinking": "low" },
151
- "browser-qa": { "model": "openai-codex/gpt-5.6-luna", "fallbackModels": ["antigravity/gemini-3-flash-preview", "zai/glm-5.3"], "thinking": "low" },
155
+ "browser-qa": { "model": "zai/glm-5.3-flash", "fallbackModels": ["openai-codex/gpt-5.6-luna"], "thinking": "low" },
152
156
  "tests": { "model": "antigravity/antigravity-claude-sonnet-4-6", "fallbackModels": ["openai-codex/gpt-5.6-luna", "zai/glm-5-turbo"], "thinking": "high" },
153
157
  "review": { "model": "antigravity/antigravity-claude-sonnet-4-6", "fallbackModels": ["openai-codex/gpt-5.6-sol", "zai/glm-5.3"], "thinking": "high" },
154
158
  "implement": { "model": "openai-codex/gpt-5.6-sol", "fallbackModels": ["zai/glm-5.3"], "thinking": "high" },
@@ -203,8 +207,8 @@
203
207
 
204
208
  "browser-qa": {
205
209
  "description": "Use for browser-based visual QA: reproduce UI bugs and verify fixes with deterministic assertions, screenshots, video, and traces.",
206
- "model": "openai-codex/gpt-5.6-luna",
207
- "fallbackModels": ["antigravity/gemini-3-flash-preview", "zai/glm-5.3"],
210
+ "model": "zai/glm-5.3-flash",
211
+ "fallbackModels": ["openai-codex/gpt-5.6-luna"],
208
212
  "thinking": "low",
209
213
  "timeoutMs": 120000,
210
214
  "tools": ["read", "grep", "bash"]
@@ -215,8 +215,8 @@ const BUILTIN_CONFIG: SubagentConfig = {
215
215
  },
216
216
  "browser-qa": {
217
217
  description: "Use for browser-based visual QA: reproduce UI bugs and verify fixes with deterministic assertions, screenshots, video, and traces.",
218
- model: "openai-codex/gpt-5.6-luna",
219
- fallbackModels: ["antigravity/gemini-3-flash-preview", "zai/glm-5.3"],
218
+ model: "zai/glm-5.3-flash",
219
+ fallbackModels: ["openai-codex/gpt-5.6-luna"],
220
220
  thinking: "low",
221
221
  timeoutMs: 120_000,
222
222
  tools: ["read", "grep", "bash"],
@@ -88,15 +88,18 @@ export function spawnAgent(
88
88
  // detached subprocesses do not depend on persisted local pi settings.
89
89
  const persistSessions = shouldPersistSubagentSessions();
90
90
  const sessionDir = persistSessions ? getAgentSessionDir(agentDir) : undefined;
91
+ const forwardedExtraArgs = options.isolatedSkills?.length ? withoutSkillArgs(extraArgs) : extraArgs;
91
92
  if (sessionDir) fs.mkdirSync(sessionDir, { recursive: true });
92
93
  const piArgs: string[] = ["--mode", "rpc"];
93
94
  if (sessionDir) piArgs.push("--session-dir", sessionDir);
94
95
  else piArgs.push("--no-session");
95
96
  piArgs.push("--no-extensions");
96
97
  piArgs.push("--extension", getModelToolsExtensionPath());
97
- // `--no-extensions` keeps sub-agents isolated, but the suite-owned provider
98
- // is infrastructure: always restore it, regardless of the selected model.
99
- piArgs.push("--extension", getAntigravityAuthExtensionPath());
98
+ // Preserve `--no-extensions` unless this invocation explicitly selects an
99
+ // Antigravity model. Environment/default models do not opt the provider in.
100
+ if (usesAntigravityModel(task.model, forwardedExtraArgs)) {
101
+ piArgs.push("--extension", getAntigravityAuthExtensionPath());
102
+ }
100
103
  if (options.isolatedSkills && options.isolatedSkills.length > 0) {
101
104
  piArgs.push("--no-skills");
102
105
  for (const skillPath of options.isolatedSkills) piArgs.push("--skill", skillPath);
@@ -111,7 +114,7 @@ export function spawnAgent(
111
114
  if (task.thinking) piArgs.push("--thinking", task.thinking);
112
115
 
113
116
  // User-supplied extra args (e.g. --thinking high)
114
- piArgs.push(...(options.isolatedSkills?.length ? withoutSkillArgs(extraArgs) : extraArgs));
117
+ piArgs.push(...forwardedExtraArgs);
115
118
  // Keep recursive/interactive parent-only tools disabled even if explicit
116
119
  // sub-agent extraArgs load additional extensions or override --tools.
117
120
  piArgs.push("--extension", getSubagentToolGuardExtensionPath());
@@ -470,6 +473,22 @@ function withoutSkillArgs(args: string[]): string[] {
470
473
  return filtered;
471
474
  }
472
475
 
476
+ function usesAntigravityModel(taskModel: string | undefined, extraArgs: string[]): boolean {
477
+ let model = taskModel?.trim() || undefined;
478
+ for (let index = 0; index < extraArgs.length; index += 1) {
479
+ const arg = extraArgs[index];
480
+ if (arg === "--model" || arg === "-m") {
481
+ model = extraArgs[index + 1]?.trim() || undefined;
482
+ index += 1;
483
+ continue;
484
+ }
485
+ if (arg.startsWith("--model=")) {
486
+ model = arg.slice("--model=".length).trim() || undefined;
487
+ }
488
+ }
489
+ return Boolean(model?.startsWith("antigravity/") && model.length > "antigravity/".length);
490
+ }
491
+
473
492
  function getModelToolsExtensionPath(): string {
474
493
  return path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..", "model-tools", "index.ts");
475
494
  }
@@ -13,6 +13,15 @@ CLI, create shared/default browser sessions, or generate executable browser code
13
13
  Never read, print, grep, copy, or edit credential values from
14
14
  `.pi/qa_auth.jsonc` yourself.
15
15
 
16
+ Treat the launch brief as a user-visible acceptance contract, not an execution
17
+ plan. It should identify the actual target URL/app when known, the user flow,
18
+ the expected observable result, and required evidence. Missing details are
19
+ preflight unknowns, not permission to invent a different target. Never create,
20
+ serve, or switch to a mock/synthetic page, test fixture, or built-in harness
21
+ unless the user explicitly requested that exact target. Source inspection and
22
+ repository tests may support discovery, but they never substitute for testing
23
+ the requested target in the browser.
24
+
16
25
  ## Workflow
17
26
 
18
27
  Treat target discovery as a 30-second preflight and invoke the runner within 45
@@ -27,9 +36,11 @@ source reading, server polling, capability probing, or retries.
27
36
  The launcher creates its private `flows/` directory and the runner rejects
28
37
  flows or evidence destinations outside this owning sub-agent directory. Do
29
38
  not override `PI_SUBAGENT_AGENT_DIR` or copy evidence to shared project paths.
30
- 3. Discover the requested target, expected behavior, and the smallest scenario
31
- that can prove it. If the target cannot be reached or started, report the
32
- concrete blocker instead of substituting static checks for browser QA.
39
+ 3. Discover the requested actual target, expected behavior, and the smallest
40
+ scenario that can prove it. Treat parent-supplied repository details as hints
41
+ unless the user explicitly requested that exact harness. If the target cannot
42
+ be reached or started, report the concrete blocker instead of switching to a
43
+ mock target or substituting static checks for browser QA.
33
44
  4. If the requested behavior requires authentication, run
34
45
  `node <runner> profiles`. A missing auth config is valid and returns an empty
35
46
  list without creating `.pi/qa_auth.jsonc`. Otherwise skip profile discovery
@@ -38,7 +49,8 @@ source reading, server polling, capability probing, or retries.
38
49
  the requested page requires login. If form authentication is required but
39
50
  there is no usable profile, follow **Form-auth scaffolding** below instead of
40
51
  asking the user to discover selectors.
41
- 5. Inspect the target code and write a declarative JSONC flow under
52
+ 5. Inspect only enough target code to identify a supported launch path or stable
53
+ locators, then write a declarative JSONC flow under
42
54
  `$PI_SUBAGENT_AGENT_DIR/browser-qa/flows/`. Never put credentials or raw
43
55
  executable JavaScript in it. The `evaluate` action exposes only the safe
44
56
  operations documented below; it does not accept expressions or scripts.
@@ -71,6 +83,23 @@ source reading, server polling, capability probing, or retries.
71
83
  and `waitForTimeout` only for short input settling or an unavoidable
72
84
  animation/debounce. Set flow `timeoutMs` only as high as the target
73
85
  legitimately needs.
86
+ - The runner waits after every visible interaction until the document is ready,
87
+ requests started by the interaction have finished, and common visible busy
88
+ markers (including `aria-busy`, progress bars, loading/spinner/skeleton test
89
+ ids and classes) disappear. It then keeps the stable state on video for 500
90
+ ms. A busy page that does not settle within `timeoutMs` fails instead of
91
+ continuing against a skeleton. For an app-specific loader not covered by
92
+ those conventions, add an explicit `waitFor`/`assertHidden` for that loader
93
+ and assert the loaded content before interacting with it.
94
+ - Recorded pointer actions are annotated automatically: clicks and double-clicks
95
+ show a cursor and pulse. After a native `dragTo` completes, the runner visibly
96
+ replays the resolved source-to-target route for 450 ms with a large orange
97
+ cursor and progressively drawn high-contrast trail, then shows a green drop
98
+ marker. The
99
+ isolated annotation layer applies to the main page, same-origin frames,
100
+ declared popups, and form-auth submission; it is accessibility-hidden, ignores
101
+ pointer input, and clears before the runner's post-action stable interval
102
+ completes.
74
103
  - Place `authRejectedIf` immediately after navigation or any transition that
75
104
  may reveal expired authentication.
76
105
  - Never weaken an assertion merely to make a failing run pass. If the observed
@@ -114,6 +143,13 @@ The top-level `environment` may set `locale`, `timezoneId`, `colorScheme`, and
114
143
  motion accepts `reduce` or `no-preference`. The resolved environment is returned
115
144
  in the result alongside the viewport.
116
145
 
146
+ Navigation and visible interaction actions automatically wait for page
147
+ readiness and a 500 ms stable recording interval. This applies to the main
148
+ page, same-origin frames and declared popups, and to the trusted form-auth
149
+ sequence. An explicit `waitForTimeout` is not extended by another automatic
150
+ delay. Click-like actions use a short bounded press duration so their automatic
151
+ video pulse remains visible even when the click immediately navigates.
152
+
117
153
  Triggering interactions (`click`, `doubleClick`, `press`, `check`, `uncheck`,
118
154
  and `selectOption`) may declare race-free expectations that are armed before
119
155
  the interaction:
@@ -41,6 +41,15 @@ Runner interactions inherit Playwright auto-waiting. Usually an action followed
41
41
  by an assertion is enough. Use `waitFor` only when the next operation depends on
42
42
  a distinct attached/detached/visible/hidden transition.
43
43
 
44
+ After navigation and visible interactions, the runner also waits for DOM
45
+ readiness, completion of requests started by that action, disappearance of
46
+ common visible `aria-busy`/progress/loading/spinner/skeleton markers, and a
47
+ 500 ms stable interval. If those signals remain busy through the flow timeout,
48
+ the run fails rather than interacting with a loading shell. This is a safe
49
+ baseline, not an application-specific oracle: explicitly wait for a custom
50
+ loader to become hidden and assert the loaded content when the application uses
51
+ different readiness semantics.
52
+
44
53
  `waitForTimeout` is bounded to five seconds and should be exceptional—for a
45
54
  known animation, debounce, or externally scheduled transition with no
46
55
  observable intermediate state. Sleeping longer hides races instead of proving
@@ -155,6 +164,16 @@ Use evidence by purpose:
155
164
  - **Video:** chronological confirmation of the complete user flow.
156
165
  - **Trace:** action/DOM timing diagnosis for a failed or flaky interaction.
157
166
 
167
+ Videos automatically show a transient cursor and yellow pulse for clicks and
168
+ double-clicks. After a native `dragTo` gesture completes, its resolved
169
+ source-to-target route is replayed over 450 ms with a large orange cursor and
170
+ progressively drawn high-contrast trail, followed by a green drop marker. These
171
+ annotations are
172
+ runner-owned, pointer-transparent, and accessibility-hidden; they cover the main
173
+ page, same-origin frames, declared popups, and form-auth submission. Their
174
+ bounded animations finish within the normal post-action stable interval, so
175
+ they explain the chronology without becoming screenshot or assertion oracles.
176
+
158
177
  Assertions determine pass/fail; evidence explains it. Preserve and link every
159
178
  artifact group returned on both passed and failed runs.
160
179