prompttest 1.4.0 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +95 -61
- package/TERMS.md +138 -134
- package/dist/bin/prompttest.js +374 -254
- package/dist/engine.bundle.js +172 -171
- package/dist/index.js +172 -171
- package/dist/lib/adaptive-timing.d.ts +55 -0
- package/dist/lib/adb.d.ts +1 -0
- package/dist/lib/jail-guard.d.ts +59 -0
- package/dist/lib/prompt-runner.d.ts +1 -0
- package/dist/lib/quiescence.d.ts +44 -0
- package/dist/lib/step-handlers.d.ts +2 -0
- package/docs/ARCHITECTURE.md +4 -4
- package/docs/CLI_CONTRACT.md +131 -0
- package/docs/CLI_STUDIO_CONTRACT.md +123 -0
- package/docs/PERFORMANCE_BASELINE.md +71 -0
- package/docs/PRODUCT_STATUS.md +56 -0
- package/docs/USER_MANUAL.md +6 -7
- package/package.json +10 -5
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @module adaptive-timing
|
|
3
|
+
* @description Device-Aware Adaptive Dynamic Waiting & Auto-Profiling.
|
|
4
|
+
*
|
|
5
|
+
* Automatically balances test timing between ultra-fast physical devices (USB 3.0)
|
|
6
|
+
* and resource-constrained CI emulators (software-rendered VMs in GitHub Actions).
|
|
7
|
+
*
|
|
8
|
+
* Implements exponential backoff condition-driven polling with instant short-circuiting:
|
|
9
|
+
* fast devices finish immediately upon event completion, while slow emulators receive
|
|
10
|
+
* adaptive headroom without CPU exhaustion.
|
|
11
|
+
*/
|
|
12
|
+
import type { AndroidDriver } from './adb.js';
|
|
13
|
+
export interface DeviceProfile {
|
|
14
|
+
/** Whether the target device is an emulator / virtual device */
|
|
15
|
+
isEmulator: boolean;
|
|
16
|
+
/** Whether the process is executing inside a CI / CD environment (e.g. GitHub Actions) */
|
|
17
|
+
isCI: boolean;
|
|
18
|
+
/** Calculated dynamic velocity multiplier (1.0 = real device, 1.5 = local emulator, 2.5 = CI) */
|
|
19
|
+
velocityMultiplier: number;
|
|
20
|
+
/** Default timeout ceiling for UI element detection */
|
|
21
|
+
defaultTimeoutMs: number;
|
|
22
|
+
/** Extended timeout ceiling for multi-step transitions and cold loads */
|
|
23
|
+
extendedTimeoutMs: number;
|
|
24
|
+
/** Minimum stability duration for dual-sample quiescence verification */
|
|
25
|
+
quiescenceSettleMs: number;
|
|
26
|
+
/** Initial polling interval in milliseconds */
|
|
27
|
+
initialPollMs: number;
|
|
28
|
+
/** Maximum backoff polling interval in milliseconds */
|
|
29
|
+
maxPollIntervalMs: number;
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* Probes the connected device and host runtime to construct an adaptive timing profile.
|
|
33
|
+
* Cached per session so that subsequent steps reuse the profiled characteristics.
|
|
34
|
+
*
|
|
35
|
+
* @param driver - Android driver instance
|
|
36
|
+
* @param serial - Optional target device serial
|
|
37
|
+
*/
|
|
38
|
+
export declare function profileDeviceEnvironment(driver: AndroidDriver, serial?: string): Promise<DeviceProfile>;
|
|
39
|
+
/**
|
|
40
|
+
* Resets the cached device profile (primarily used in testing).
|
|
41
|
+
*/
|
|
42
|
+
export declare function resetDeviceProfile(): void;
|
|
43
|
+
/**
|
|
44
|
+
* Polls an asynchronous condition using adaptive backoff with immediate short-circuiting.
|
|
45
|
+
*
|
|
46
|
+
* Fast devices short-circuit the instant the condition returns a truthy value.
|
|
47
|
+
* Slow emulators receive backoff polling that preserves CPU cycles for rendering.
|
|
48
|
+
*
|
|
49
|
+
* @param checkFn - Async callback returning a truthy value when the condition is met.
|
|
50
|
+
* @param timeoutMs - Maximum duration to wait before giving up.
|
|
51
|
+
* @param initialIntervalMs - Starting interval between checks.
|
|
52
|
+
* @param maxIntervalMs - Cap for the backoff interval.
|
|
53
|
+
* @returns The resolved value of the condition, or throws a timeout Error.
|
|
54
|
+
*/
|
|
55
|
+
export declare function pollWithAdaptiveBackoff<T>(checkFn: () => Promise<T | null | undefined | false>, timeoutMs: number, initialIntervalMs?: number, maxIntervalMs?: number): Promise<T>;
|
package/dist/lib/adb.d.ts
CHANGED
|
@@ -406,6 +406,7 @@ export declare class AndroidDriver {
|
|
|
406
406
|
}, localDestPath: string, serial?: string): Promise<string>;
|
|
407
407
|
/**
|
|
408
408
|
* Checks if the virtual software keyboard (IME) is currently active and visible on screen.
|
|
409
|
+
* Uses fast shell grep filtering across Android IME & Window properties with resilient fallbacks.
|
|
409
410
|
* @param serial Optional target device serial.
|
|
410
411
|
* @returns Promise resolving to true if virtual keyboard is displayed.
|
|
411
412
|
*/
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @module jail-guard
|
|
3
|
+
* @description Zero-Overhead In-Memory App Boundary Guardian for PromptTest.
|
|
4
|
+
*
|
|
5
|
+
* Enforces strict package boundary isolation to guarantee that automation
|
|
6
|
+
* NEVER sends touch, text, or gesture events into third-party personal apps
|
|
7
|
+
* (e.g. banking apps, email, browsers) or home screen launchers.
|
|
8
|
+
*
|
|
9
|
+
* Operates directly on the in-memory parsed hierarchy (UiNode.packageName)
|
|
10
|
+
* with 0 ms additional ADB overhead in normal operation.
|
|
11
|
+
*/
|
|
12
|
+
import type { UiNode } from './crawler.js';
|
|
13
|
+
import type { AndroidDriver } from './adb.js';
|
|
14
|
+
/**
|
|
15
|
+
* Registry of known Android OEM launchers and desktop interfaces.
|
|
16
|
+
*/
|
|
17
|
+
export declare const KNOWN_LAUNCHER_PACKAGES: readonly string[];
|
|
18
|
+
/**
|
|
19
|
+
* Registry of allowed Android system overlay and framework packages
|
|
20
|
+
* (keyboards, permission dialogs, autofill, package installer).
|
|
21
|
+
*/
|
|
22
|
+
export declare const KNOWN_SYSTEM_PACKAGES: readonly string[];
|
|
23
|
+
/**
|
|
24
|
+
* Checks if a given package name corresponds to a known launcher or desktop interface.
|
|
25
|
+
*/
|
|
26
|
+
export declare function isLauncherPackage(pkg: string): boolean;
|
|
27
|
+
/**
|
|
28
|
+
* Checks if a given package name corresponds to an allowed Android system overlay.
|
|
29
|
+
*/
|
|
30
|
+
export declare function isAllowedSystemPackage(pkg: string): boolean;
|
|
31
|
+
/**
|
|
32
|
+
* Determines whether a specific node's package belongs to the target app under test
|
|
33
|
+
* or an allowed system overlay (keyboard, permission prompt, system UI).
|
|
34
|
+
*
|
|
35
|
+
* @param nodePackage - The package name of the UI node being interacted with.
|
|
36
|
+
* @param targetPackage - The target application package under test (if specified).
|
|
37
|
+
*/
|
|
38
|
+
export declare function isTargetOrAllowedSystem(nodePackage?: string, targetPackage?: string): boolean;
|
|
39
|
+
/**
|
|
40
|
+
* Extracts the primary / foreground package from a flattened list of UiNodes.
|
|
41
|
+
* Evaluates non-system, interactive nodes first, then falls back to the root node.
|
|
42
|
+
*/
|
|
43
|
+
export declare function detectScreenPackage(nodes: UiNode[]): string | null;
|
|
44
|
+
/**
|
|
45
|
+
* Guards the app boundary: If an escape is detected (screen belongs to a launcher or foreign app),
|
|
46
|
+
* it blocks further actions, prevents unsafe back-presses on launchers, and brings the target app
|
|
47
|
+
* back to the foreground.
|
|
48
|
+
*
|
|
49
|
+
* @param nodes - In-memory UI nodes from current hierarchy dump.
|
|
50
|
+
* @param targetPackage - The expected target application package.
|
|
51
|
+
* @param driver - Android driver instance for containment recovery.
|
|
52
|
+
* @param serial - Optional target device serial.
|
|
53
|
+
* @returns Details on whether an escape occurred and whether recovery succeeded.
|
|
54
|
+
*/
|
|
55
|
+
export declare function guardAppBoundary(nodes: UiNode[], targetPackage: string, driver: AndroidDriver, serial?: string): Promise<{
|
|
56
|
+
escaped: boolean;
|
|
57
|
+
recovered: boolean;
|
|
58
|
+
currentPackage: string | null;
|
|
59
|
+
}>;
|
|
@@ -19,6 +19,7 @@ export declare class PromptRunner {
|
|
|
19
19
|
private lastHierarchy;
|
|
20
20
|
private screenDimensions;
|
|
21
21
|
private navBarHeight;
|
|
22
|
+
private targetPackage;
|
|
22
23
|
/**
|
|
23
24
|
* Initializes a new instance of the PromptRunner.
|
|
24
25
|
* Uses dependency injection for core services, defaulting to singletons if omitted.
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @module quiescence
|
|
3
|
+
* @description Dynamic Screen Quiescence & Idling Sentinel for PromptTest.
|
|
4
|
+
*
|
|
5
|
+
* Implements self-synchronizing stability detection inspired by Playwright's networkidle
|
|
6
|
+
* and Android Espresso's IdlingResource. Replaces fragile static timers with dual-sample
|
|
7
|
+
* layout stabilization and active in-flight loading sentinel checks.
|
|
8
|
+
*/
|
|
9
|
+
import type { UiNode } from './crawler.js';
|
|
10
|
+
import type { ExecutionContext } from './step-handlers.js';
|
|
11
|
+
export interface QuiescenceResult {
|
|
12
|
+
/** Whether the screen reached verified stability before the timeout */
|
|
13
|
+
settled: boolean;
|
|
14
|
+
/** Milliseconds elapsed during stabilization */
|
|
15
|
+
durationMs: number;
|
|
16
|
+
/** Final structural hash of the quiescent screen */
|
|
17
|
+
finalHash: string;
|
|
18
|
+
/** Flattened UI nodes of the quiescent screen */
|
|
19
|
+
flat: UiNode[];
|
|
20
|
+
}
|
|
21
|
+
/**
|
|
22
|
+
* Inspects a flattened UI hierarchy to detect active in-flight network spinners,
|
|
23
|
+
* shimmers, progress bars, or server-waiting indicators.
|
|
24
|
+
*
|
|
25
|
+
* @param nodes - Flattened array of UiNodes
|
|
26
|
+
* @returns Status indicating whether loading is in progress and why.
|
|
27
|
+
*/
|
|
28
|
+
export declare function isLoadingInProgress(nodes: UiNode[]): {
|
|
29
|
+
loading: boolean;
|
|
30
|
+
reason?: string;
|
|
31
|
+
};
|
|
32
|
+
/**
|
|
33
|
+
* Dynamically waits until the application screen achieves 100% quiescence:
|
|
34
|
+
* 1. All active loading indicators and spinners have vanished.
|
|
35
|
+
* 2. Two consecutive UI hierarchy snapshots yield matching structural hashes.
|
|
36
|
+
*
|
|
37
|
+
* Exits immediately once settled, saving seconds on fast devices while providing
|
|
38
|
+
* dynamic patience for slow network requests or cloud emulators.
|
|
39
|
+
*
|
|
40
|
+
* @param ctx - Execution context providing driver and hierarchy access.
|
|
41
|
+
* @param timeoutMs - Maximum wait ceiling (defaults to 4500ms).
|
|
42
|
+
* @param sampleIntervalMs - Interval between stability probe samples (defaults to 150ms).
|
|
43
|
+
*/
|
|
44
|
+
export declare function waitForQuiescence(ctx: ExecutionContext, timeoutMs?: number, sampleIntervalMs?: number): Promise<QuiescenceResult>;
|
|
@@ -21,6 +21,8 @@ import { PromptStep } from './prompt-runner.js';
|
|
|
21
21
|
* It encapsulates the driver instance, state cache, and common UI interaction logic.
|
|
22
22
|
*/
|
|
23
23
|
export interface ExecutionContext {
|
|
24
|
+
/** Target application package name under test (used for jail guard containment). */
|
|
25
|
+
targetPackage?: string;
|
|
24
26
|
/** The ADB-based Android UI driver used to interact with the device. */
|
|
25
27
|
driver: AndroidDriver;
|
|
26
28
|
/** Memory engine used to record and learn resilient locators over time. */
|
package/docs/ARCHITECTURE.md
CHANGED
|
@@ -6,7 +6,7 @@ PromptTest is a zero-code, AI-assisted Android test automation and visual QA fra
|
|
|
6
6
|
|
|
7
7
|
## 1. High-Level Architecture Overview
|
|
8
8
|
|
|
9
|
-
PromptTest operates on a modular layered architecture where test scripts written in natural language or AST-driven directives are compiled, bound to live or synthetic data, executed against connected devices via an isolated locking layer, and reported with visual diffs and interactive HTML reports.
|
|
9
|
+
PromptTest operates on a modular layered architecture where test scripts written in natural language or AST-driven directives are compiled, bound to live or synthetic data, executed against connected devices via an isolated locking layer, and reported with visual diffs and interactive HTML reports. Current release boundaries are tracked in [PRODUCT_STATUS.md](PRODUCT_STATUS.md).
|
|
10
10
|
|
|
11
11
|
```
|
|
12
12
|
┌───────────────────────────────────────────────────────────┐
|
|
@@ -35,14 +35,14 @@ PromptTest operates on a modular layered architecture where test scripts written
|
|
|
35
35
|
│ │ │
|
|
36
36
|
┌──────▼──────────────────────▼──────────────────────▼──────┐
|
|
37
37
|
│ Visual Testing & Diagnostics │
|
|
38
|
-
│ lib/baseline.ts (
|
|
38
|
+
│ lib/baseline.ts (pixelmatch / PNG Diff) │
|
|
39
39
|
│ lib/explorer.ts (Autonomous Crawler) │
|
|
40
40
|
└─────────────────────────────┬─────────────────────────────┘
|
|
41
41
|
│
|
|
42
42
|
┌─────────────────────────────▼─────────────────────────────┐
|
|
43
43
|
│ Reporting & Observability │
|
|
44
44
|
│ lib/reporter.ts (HTML, JSON, JUnit XML, Markdown) │
|
|
45
|
-
│ lib/live-server.ts (
|
|
45
|
+
│ lib/live-server.ts (HTTP / SSE Live Monitor) │
|
|
46
46
|
└───────────────────────────────────────────────────────────┘
|
|
47
47
|
```
|
|
48
48
|
|
|
@@ -100,7 +100,7 @@ Instead of an unwieldy switch-case statement, execution is delegated to decouple
|
|
|
100
100
|
|
|
101
101
|
### 2.6 Visual Baselines & Computer Vision (`lib/baseline.ts`, `lib/explorer.ts`)
|
|
102
102
|
|
|
103
|
-
- Pixel-
|
|
103
|
+
- Pixel-level image diffing using `pixelmatch` and PNG decoding via `pngjs`.
|
|
104
104
|
- Configurable failure thresholds (e.g. `diffThreshold: 0.02` for 2% variance).
|
|
105
105
|
- Automatic bounding box masking for dynamic regions (e.g. status bar clocks, live feeds) to eliminate false positives in visual regression testing.
|
|
106
106
|
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
# PromptTest CLI Contract
|
|
2
|
+
|
|
3
|
+
**Contract date:** 2026-09-22
|
|
4
|
+
**Package:** `prompttest` 1.3.5
|
|
5
|
+
|
|
6
|
+
This document describes the currently shipped CLI and SDK integration surfaces. It is intended for CI consumers and desktop clients. It does not promise functionality that is only planned or hardware-unvalidated.
|
|
7
|
+
|
|
8
|
+
## Compatibility Rules
|
|
9
|
+
|
|
10
|
+
- Existing public exports and command behavior are additive-compatible within a major version.
|
|
11
|
+
- Consumers should use exported types and documented fields rather than parsing terminal text.
|
|
12
|
+
- Report fields may gain optional properties. Existing required fields will not be removed without a breaking release.
|
|
13
|
+
- Hardware operations require ADB and a reachable Android device or emulator.
|
|
14
|
+
|
|
15
|
+
## Exit Codes
|
|
16
|
+
|
|
17
|
+
| Code | Name | Meaning |
|
|
18
|
+
| ---: | -------------------- | ----------------------------------------------------------------- |
|
|
19
|
+
| 0 | `SUCCESS` | Run completed successfully. |
|
|
20
|
+
| 1 | `TEST_FAILED` | An assertion, locator, or test step failed. |
|
|
21
|
+
| 2 | `DEVICE_UNAVAILABLE` | ADB/device connection or device availability failed. |
|
|
22
|
+
| 3 | `CONFIG_ERROR` | A spec, configuration, argument, or include reference is invalid. |
|
|
23
|
+
| 4 | `LOCKED` | The target device is locked by another PromptTest process. |
|
|
24
|
+
|
|
25
|
+
CI integrations should use these codes instead of matching human-readable output.
|
|
26
|
+
|
|
27
|
+
## Report Model
|
|
28
|
+
|
|
29
|
+
The canonical report is JSON with this shape:
|
|
30
|
+
|
|
31
|
+
```ts
|
|
32
|
+
interface QaReportData {
|
|
33
|
+
packageName: string;
|
|
34
|
+
deviceId: string;
|
|
35
|
+
startTime: string;
|
|
36
|
+
endTime: string;
|
|
37
|
+
durationSeconds: number;
|
|
38
|
+
totalSteps: number;
|
|
39
|
+
passedSteps: number;
|
|
40
|
+
failedSteps: number;
|
|
41
|
+
skippedSteps: number;
|
|
42
|
+
warnSteps: number;
|
|
43
|
+
errorsDetected: string[];
|
|
44
|
+
videoPath?: string;
|
|
45
|
+
steps: AuditStep[];
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
interface AuditStep {
|
|
49
|
+
stepIndex: number;
|
|
50
|
+
timestamp: string;
|
|
51
|
+
phase: string;
|
|
52
|
+
description: string;
|
|
53
|
+
action: string;
|
|
54
|
+
status: 'PASS' | 'WARN' | 'FAIL' | 'SKIPPED';
|
|
55
|
+
target?: string;
|
|
56
|
+
screenshotPath?: string;
|
|
57
|
+
durationMs?: number;
|
|
58
|
+
triageBundle?: string;
|
|
59
|
+
videoOffsetMs?: number;
|
|
60
|
+
details?: Record<string, unknown>;
|
|
61
|
+
}
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Supported serializers are JSON, Markdown, JUnit XML, and HTML. JSON is the machine-readable source of truth; the other formats are views of the same run.
|
|
65
|
+
|
|
66
|
+
## Output Conventions
|
|
67
|
+
|
|
68
|
+
The default output directory is `output/` in the working directory. Common artifacts include:
|
|
69
|
+
|
|
70
|
+
- `<name>-results.json`
|
|
71
|
+
- `<name>-report.md`
|
|
72
|
+
- `<name>-report.html`
|
|
73
|
+
- `<name>-junit.xml`
|
|
74
|
+
- Failure triage folders or ZIPs containing screenshots, hierarchy XML, logcat, and match scores
|
|
75
|
+
- Baselines under `output/baselines/`
|
|
76
|
+
- Checkpoints under `output/checkpoints/`
|
|
77
|
+
- Optional MP4 recordings referenced by `videoPath`
|
|
78
|
+
|
|
79
|
+
Paths in report fields are relative to the report/output directory unless the producer explicitly returns an absolute path.
|
|
80
|
+
|
|
81
|
+
## Live Engine Protocol
|
|
82
|
+
|
|
83
|
+
The live server is started with `prompttest serve [port]` or `startLiveServer()`. Defaults:
|
|
84
|
+
|
|
85
|
+
- Host: `127.0.0.1`
|
|
86
|
+
- Port: `4040`
|
|
87
|
+
- Transport: HTTP plus Server-Sent Events
|
|
88
|
+
- Event stream: `GET /events`
|
|
89
|
+
- Report view: `GET /` or `GET /report`
|
|
90
|
+
|
|
91
|
+
The current server has no authentication and sends permissive CORS headers. It must be treated as a local trusted-process interface. Do not bind it to a non-loopback host or expose it outside a controlled machine without adding authentication and access controls.
|
|
92
|
+
|
|
93
|
+
### Stable read endpoints
|
|
94
|
+
|
|
95
|
+
- `GET /api/devices` returns connected device metadata.
|
|
96
|
+
- `GET /api/screen-state?deviceId=<id>&hierarchy=true|false` returns a screenshot, foreground package, and optional flattened UI elements.
|
|
97
|
+
- `GET /api/specs` lists local `.txt` and `.spec` files.
|
|
98
|
+
- `GET /api/specs/<name>` reads a spec file.
|
|
99
|
+
- `GET /api/reports` lists generated reports.
|
|
100
|
+
- `GET /api/reports/<name>` reads a report detail.
|
|
101
|
+
- `GET /api/baselines` lists available baselines.
|
|
102
|
+
- `GET /api/baselines/masks` lists baseline masks.
|
|
103
|
+
- `GET /api/report` and `GET /results.json` return the current report data.
|
|
104
|
+
|
|
105
|
+
### Mutating endpoints
|
|
106
|
+
|
|
107
|
+
These endpoints control a connected device or write local files and should only be called by an explicit local user action:
|
|
108
|
+
|
|
109
|
+
- `POST /api/tap` with `{ x, y, deviceId? }`
|
|
110
|
+
- `POST /api/type` with `{ text, deviceId? }`
|
|
111
|
+
- `POST /api/key` with `{ keyCode, deviceId? }`
|
|
112
|
+
- `POST /api/wake` with `{ deviceId? }`
|
|
113
|
+
- `POST /api/specs` with `{ name, content }`
|
|
114
|
+
- `POST /api/action` with `{ step, deviceId? }`
|
|
115
|
+
- `POST /api/run` with `{ specName? | spec?, deviceId?, saveBaseline?, compareBaseline? }`
|
|
116
|
+
- Baseline approval, comparison, upload, mask, clear, and delete endpoints under `/api/baselines/*`
|
|
117
|
+
|
|
118
|
+
### SSE events
|
|
119
|
+
|
|
120
|
+
Clients connecting to `/events` may receive:
|
|
121
|
+
|
|
122
|
+
- `step`: step number, instruction, status, duration, and optional error
|
|
123
|
+
- `complete`: final `QaReportData`
|
|
124
|
+
- `error`: an error message payload
|
|
125
|
+
- Heartbeat comments every 15 seconds
|
|
126
|
+
|
|
127
|
+
Clients must reconnect and tolerate duplicate or missing transient events; the final report is authoritative.
|
|
128
|
+
|
|
129
|
+
## Studio Boundary
|
|
130
|
+
|
|
131
|
+
The CLI owns device control, test execution, report generation, and artifact storage. A Studio client owns presentation, browsing, device selection, and desktop packaging. Studio must not duplicate locator resolution or report aggregation logic. Any new endpoint or event should be added here with a compatibility note and an integration test.
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
# CLI and Studio Contract
|
|
2
|
+
|
|
3
|
+
**Contract version:** `1`
|
|
4
|
+
**CLI package:** `prompttest` 1.3.5
|
|
5
|
+
**Status:** Current implementation contract, verified 2026-09-22
|
|
6
|
+
|
|
7
|
+
This document defines the integration surface between the PromptTest CLI engine and a desktop or browser Studio client. The CLI remains independently usable; Studio consumes this contract and must not reimplement device control, test execution, report generation, or baseline comparison.
|
|
8
|
+
|
|
9
|
+
## Engine Startup
|
|
10
|
+
|
|
11
|
+
Start the engine with:
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
prompttest serve [port]
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
The default port is `4040`. The server binds to `127.0.0.1` by default. Programmatic consumers can call `startLiveServer({ port, host, outputDir, specsDir })`.
|
|
18
|
+
|
|
19
|
+
The server is not an authenticated remote control plane. It exposes device actions, screenshots, UI hierarchy data, specifications, reports, and baselines. Keep the default loopback binding; do not expose the server on a network interface without adding authentication and authorization first.
|
|
20
|
+
|
|
21
|
+
## Exit Codes
|
|
22
|
+
|
|
23
|
+
| Code | Meaning |
|
|
24
|
+
| ---: | --------------------------------------------- |
|
|
25
|
+
| `0` | Successful execution |
|
|
26
|
+
| `1` | Test assertion or locator failure |
|
|
27
|
+
| `2` | Device unavailable, offline, or unauthorized |
|
|
28
|
+
| `3` | Invalid configuration, spec, or CLI arguments |
|
|
29
|
+
| `4` | Device lock contention |
|
|
30
|
+
|
|
31
|
+
Studio should display the error message and preserve the numeric exit code when it starts the CLI as a child process.
|
|
32
|
+
|
|
33
|
+
## Report Contract
|
|
34
|
+
|
|
35
|
+
The canonical report is JSON. The current `QaReportData` shape is:
|
|
36
|
+
|
|
37
|
+
```json
|
|
38
|
+
{
|
|
39
|
+
"packageName": "com.example.app",
|
|
40
|
+
"deviceId": "emulator-5554",
|
|
41
|
+
"startTime": "2026-09-22T00:00:00.000Z",
|
|
42
|
+
"endTime": "2026-09-22T00:00:03.000Z",
|
|
43
|
+
"durationSeconds": 3,
|
|
44
|
+
"totalSteps": 2,
|
|
45
|
+
"passedSteps": 1,
|
|
46
|
+
"failedSteps": 1,
|
|
47
|
+
"skippedSteps": 0,
|
|
48
|
+
"warnSteps": 0,
|
|
49
|
+
"errorsDetected": [],
|
|
50
|
+
"videoPath": "login-recording.mp4",
|
|
51
|
+
"steps": []
|
|
52
|
+
}
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Each step can contain `stepIndex`, `timestamp`, `phase`, `description`, `action`, `target`, `status`, `screenshotPath`, `durationMs`, `triageBundle`, `videoOffsetMs`, and `details`.
|
|
56
|
+
|
|
57
|
+
Supported report serializers are `json`, `md`, `junit`, and `html`. JSON is the integration source of truth. File names use the selected base name:
|
|
58
|
+
|
|
59
|
+
- `<name>-results.json`
|
|
60
|
+
- `<name>-report.md`
|
|
61
|
+
- `<name>-junit.xml`
|
|
62
|
+
- `<name>-report.html`
|
|
63
|
+
|
|
64
|
+
## SSE Events
|
|
65
|
+
|
|
66
|
+
Connect to `GET /events` with an `EventSource`. Events currently emitted are:
|
|
67
|
+
|
|
68
|
+
- `step`: step number, instruction, status, duration, and optional error
|
|
69
|
+
- `complete`: final `QaReportData`
|
|
70
|
+
- `error`: an error message object
|
|
71
|
+
|
|
72
|
+
Clients should tolerate unknown future events and reconnect after a dropped connection. The server sends comment heartbeats approximately every 15 seconds.
|
|
73
|
+
|
|
74
|
+
## HTTP Routes
|
|
75
|
+
|
|
76
|
+
All successful JSON responses include `success: true` where applicable. Errors use `success: false` and an `error` string.
|
|
77
|
+
|
|
78
|
+
### Devices and live device state
|
|
79
|
+
|
|
80
|
+
- `GET /api/devices`: connected device list
|
|
81
|
+
- `GET /api/screen-state?deviceId=<id>&hierarchy=true|false`: screenshot, foreground package, and UI elements
|
|
82
|
+
- `POST /api/tap`: `{ "x": number, "y": number, "deviceId"?: string }`
|
|
83
|
+
- `POST /api/type`: `{ "text": string, "deviceId"?: string }`
|
|
84
|
+
- `POST /api/key`: `{ "keyCode": number|string, "deviceId"?: string }`
|
|
85
|
+
- `POST /api/wake`: `{ "deviceId"?: string }`
|
|
86
|
+
|
|
87
|
+
### Specifications and execution
|
|
88
|
+
|
|
89
|
+
- `GET /api/specs`: list `.txt` and `.spec` files
|
|
90
|
+
- `GET /api/specs/<name>` or `GET /api/spec?name=<name>`: read a specification
|
|
91
|
+
- `POST /api/specs` or `POST /api/spec`: `{ "name": string, "content": string }`
|
|
92
|
+
- `POST /api/action`: `{ "step": string, "deviceId"?: string }`
|
|
93
|
+
- `POST /api/run`: `{ "specName"?: string, "spec"?: string, "deviceId"?: string, "saveBaseline"?: boolean, "compareBaseline"?: boolean }`
|
|
94
|
+
|
|
95
|
+
`/api/run` starts execution asynchronously. Use SSE and report endpoints to observe completion.
|
|
96
|
+
|
|
97
|
+
### Reports and baselines
|
|
98
|
+
|
|
99
|
+
- `GET /api/reports`: list available report artifacts
|
|
100
|
+
- `GET /api/reports/<name>` or `GET /api/report-detail?name=<name>`: retrieve report details
|
|
101
|
+
- `GET /api/report` or `GET /results.json`: retrieve the current JSON report
|
|
102
|
+
- `GET /api/baselines`: list baseline images and metadata
|
|
103
|
+
- `POST /api/baselines/compare`: compare an image against a baseline
|
|
104
|
+
- `POST /api/baselines/approve`: approve a baseline result
|
|
105
|
+
- `POST /api/baselines/clear`: clear baseline data
|
|
106
|
+
- `POST /api/baselines/delete-step`: remove a step baseline
|
|
107
|
+
- `POST /api/baselines/upload`: upload a baseline image
|
|
108
|
+
- `GET /api/baselines/masks`: list baseline masks
|
|
109
|
+
- `POST /api/baselines/masks`: create or update a baseline mask
|
|
110
|
+
|
|
111
|
+
The exact request bodies for baseline management should be generated from the CLI client implementation until a schema is added in a future contract version.
|
|
112
|
+
|
|
113
|
+
## Artifact Rules
|
|
114
|
+
|
|
115
|
+
The CLI owns artifact creation under `output/` by default. Studio should display and download paths returned by reports rather than copying large screenshots or videos into JSON. Triage artifacts may include screenshots, hierarchy XML, logcat, match scores, a manifest, and a ZIP archive. A video path and failure offset do not imply that a clipped video snippet exists.
|
|
116
|
+
|
|
117
|
+
## Compatibility Rules
|
|
118
|
+
|
|
119
|
+
- Existing public SDK exports and CLI flags are additive and semver-protected.
|
|
120
|
+
- Studio must tolerate missing optional fields and unknown JSON fields.
|
|
121
|
+
- Studio must not assume Android hardware is available during engine startup.
|
|
122
|
+
- A failed run is not an engine crash; inspect the report and exit code separately.
|
|
123
|
+
- Contract changes require a version update, documentation update, and package smoke-test coverage.
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
# PromptTest Performance Baseline
|
|
2
|
+
|
|
3
|
+
**Measured:** 2026-09-22
|
|
4
|
+
**Package:** `prompttest` 1.3.5
|
|
5
|
+
**Environment:** macOS, Node.js 24.13.0, no ADB device required
|
|
6
|
+
|
|
7
|
+
This is a reproducible CI and package-size baseline. Synthetic timings are health signals, not physical-device performance claims.
|
|
8
|
+
|
|
9
|
+
## Current Measurements
|
|
10
|
+
|
|
11
|
+
- Offline test suite: 39 test files, 337 tests passing
|
|
12
|
+
- Test wall time: approximately 14 seconds in the current environment
|
|
13
|
+
- CLI bundle: approximately 332 KB
|
|
14
|
+
- SDK bundle: approximately 316 KB
|
|
15
|
+
- Engine bundle: approximately 316 KB
|
|
16
|
+
- Runtime dependencies: `chalk`, `pixelmatch`, `pngjs`
|
|
17
|
+
- Supported Node.js range: `>=18.0.0`
|
|
18
|
+
|
|
19
|
+
## Existing Benchmark Coverage
|
|
20
|
+
|
|
21
|
+
`tests/performance-benchmark.test.ts` and `lib/benchmark.ts` cover:
|
|
22
|
+
|
|
23
|
+
- Synthetic screenshot allocation timing
|
|
24
|
+
- Synthetic UI hierarchy timing
|
|
25
|
+
- Matching latency across a synthetic 1,000-node hierarchy
|
|
26
|
+
- Percentile calculation and summary metrics
|
|
27
|
+
- Hierarchy hashing and structural diffing
|
|
28
|
+
- Verify-handler cache reuse
|
|
29
|
+
- Reporter screenshot-buffer cleanup
|
|
30
|
+
- PromptRunner benchmark mode
|
|
31
|
+
|
|
32
|
+
Run the offline benchmark tests with:
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
npx vitest run tests/performance-benchmark.test.ts
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Run the complete offline suite with:
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
npm test
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Required Hardware Measurements
|
|
45
|
+
|
|
46
|
+
The following values must be collected on representative Android emulator and physical-device configurations before making performance promises:
|
|
47
|
+
|
|
48
|
+
- ADB screenshot p50/p95 latency
|
|
49
|
+
- `uiautomator dump` p50/p95 latency
|
|
50
|
+
- Hierarchy dumps per visited screen
|
|
51
|
+
- Screenshots per exploration state
|
|
52
|
+
- Peak RSS for a 50-screen exploration
|
|
53
|
+
- HTML report size with and without embedded screenshots
|
|
54
|
+
- Triage ZIP size
|
|
55
|
+
- Aggregation time for large output directories
|
|
56
|
+
- Video-enabled versus video-disabled run time
|
|
57
|
+
|
|
58
|
+
Hardware results should include device model, Android version, screen size, Node version, iteration count, and whether the device is USB, Wi-Fi, or emulator. Do not compare synthetic benchmark timings directly with ADB measurements.
|
|
59
|
+
|
|
60
|
+
## Release Guardrails
|
|
61
|
+
|
|
62
|
+
Any new runtime dependency or report-embedding behavior should be evaluated against:
|
|
63
|
+
|
|
64
|
+
- Published tarball size
|
|
65
|
+
- Installed unpacked size
|
|
66
|
+
- Cold CLI startup time
|
|
67
|
+
- Peak memory during report generation
|
|
68
|
+
- Existing offline test duration
|
|
69
|
+
- Node 18 compatibility
|
|
70
|
+
|
|
71
|
+
The npm package should remain independently usable without Studio or a running server.
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# PromptTest Product Status
|
|
2
|
+
|
|
3
|
+
**Status date:** 2026-09-22
|
|
4
|
+
**Package:** `prompttest` 1.3.5
|
|
5
|
+
|
|
6
|
+
This document is the current status index for the PromptTest CLI. Older implementation plans and the internal SRS are historical context, not release requirements.
|
|
7
|
+
|
|
8
|
+
## Status Definitions
|
|
9
|
+
|
|
10
|
+
- **Released:** Implemented, tested in the repository, and included in the published package.
|
|
11
|
+
- **Implemented:** Present in the CLI, but physical-device or cross-platform validation may still be incomplete.
|
|
12
|
+
- **Planned:** Intended work that is not part of the current release contract.
|
|
13
|
+
- **Studio:** Requires the separate desktop application; that source is not part of this repository.
|
|
14
|
+
|
|
15
|
+
## Current CLI Contract
|
|
16
|
+
|
|
17
|
+
### Released or implemented
|
|
18
|
+
|
|
19
|
+
- Android execution through ADB
|
|
20
|
+
- Natural-language prompt parsing and structured flow control
|
|
21
|
+
- Device locking and device-pool support
|
|
22
|
+
- Data-driven runs and dynamic test data
|
|
23
|
+
- Visual baselines using `pixelmatch` and PNG decoding
|
|
24
|
+
- Markdown, JSON, JUnit, and HTML reporting
|
|
25
|
+
- Failure triage artifacts and optional video recording
|
|
26
|
+
- Autonomous exploration with DFS traversal and safety modes
|
|
27
|
+
- Optional AI providers, disabled unless configured
|
|
28
|
+
- HTTP/SSE live monitoring server
|
|
29
|
+
- CI workflow generation through `init-ci`
|
|
30
|
+
- Committed offline quality CI and scheduled emulator CI workflows
|
|
31
|
+
- npm package and standalone distribution build paths
|
|
32
|
+
|
|
33
|
+
### Implemented with validation limits
|
|
34
|
+
|
|
35
|
+
- iOS driver interfaces require external Apple tooling and need a documented real-device acceptance matrix.
|
|
36
|
+
- Autonomous exploration is unit-tested and fixture-tested; coverage across physical applications is not implied.
|
|
37
|
+
- The emulator workflow is configured but requires a successful GitHub Actions run before hardware support is release-validated.
|
|
38
|
+
- Live-server consumers must use the documented loopback and artifact-access contract.
|
|
39
|
+
- Hardware tests require ADB, an emulator, or a connected device and are not part of the offline default test guarantee.
|
|
40
|
+
|
|
41
|
+
### Planned or deferred
|
|
42
|
+
|
|
43
|
+
- Performance benchmark evidence and output-retention limits
|
|
44
|
+
- Stronger subprocess isolation for untrusted JavaScript specifications
|
|
45
|
+
- Locale packs and broader non-English resolution
|
|
46
|
+
- Spec scaffolding and a dedicated profiler
|
|
47
|
+
- Offline licensing and report watermarking
|
|
48
|
+
- Hosted collaboration, cloud reporting, and multi-device orchestration
|
|
49
|
+
|
|
50
|
+
## CLI and Studio Boundary
|
|
51
|
+
|
|
52
|
+
The CLI owns device control, execution, comparison, report generation, and artifact contracts. A separate Studio application may consume those contracts for device selection, live views, report browsing, packaging, and desktop workflows. Studio release status must be tracked in its own repository and must not be represented as CLI completion.
|
|
53
|
+
|
|
54
|
+
## npm Compatibility Policy
|
|
55
|
+
|
|
56
|
+
Changes to the CLI must preserve existing public exports and behavior unless a versioned breaking change is explicitly planned. Every release should pass typecheck, lint, offline tests, package verification, tarball installation smoke tests, and the supported Node.js compatibility checks before publication.
|
package/docs/USER_MANUAL.md
CHANGED
|
@@ -584,7 +584,7 @@ PromptTest can capture visual baselines and perform perceptual diffing to detect
|
|
|
584
584
|
|
|
585
585
|
- `--save-baseline`: Saves the current screen layout as a visual baseline for future comparisons.
|
|
586
586
|
- `--compare-baseline`: Compares the current screen layout against the saved baseline and fails if differences exceed the threshold.
|
|
587
|
-
- `--baseline-threshold=N`: Adjusts the
|
|
587
|
+
- `--baseline-threshold=N`: Adjusts the diffing tolerance as a ratio from `0` to `1` (e.g., `--baseline-threshold=0.05` for 5%).
|
|
588
588
|
|
|
589
589
|
### Custom Reporters
|
|
590
590
|
|
|
@@ -615,7 +615,7 @@ PromptTest includes a cross-process, file-based mutex lock (`lib/lock.ts`) that
|
|
|
615
615
|
PromptTest exposes a clean programmatic API using factory functions and a unified `DriverInterface`. Both Android and iOS drivers share the identical driver abstraction:
|
|
616
616
|
|
|
617
617
|
```typescript
|
|
618
|
-
import { createPromptRunner, createAndroidDriver,
|
|
618
|
+
import { createPromptRunner, createAndroidDriver, IosDriver } from 'prompttest';
|
|
619
619
|
|
|
620
620
|
async function runCrossPlatformTests() {
|
|
621
621
|
// 🤖 Android Execution
|
|
@@ -624,9 +624,8 @@ async function runCrossPlatformTests() {
|
|
|
624
624
|
const androidReport = await androidRunner.runSpec('specs/login.txt', 'com.example.app');
|
|
625
625
|
|
|
626
626
|
// 🍎 iOS Execution (Simulator or Physical Device)
|
|
627
|
-
const iosDriver =
|
|
627
|
+
const iosDriver = new IosDriver({
|
|
628
628
|
udid: '00008101-001234567890',
|
|
629
|
-
bundleId: 'com.example.app',
|
|
630
629
|
});
|
|
631
630
|
const iosRunner = createPromptRunner(iosDriver);
|
|
632
631
|
const iosReport = await iosRunner.runSpec('specs/login.txt', 'com.example.app');
|
|
@@ -760,6 +759,6 @@ PromptTest is a **fast UI-tree automation engine over ADB**. It "sees" what `uia
|
|
|
760
759
|
|
|
761
760
|
## 💡 Support & Contribution
|
|
762
761
|
|
|
763
|
-
- **Repository**: [github.com/shriramsingh/
|
|
764
|
-
- **Issues & Requests**: [File an Issue](https://github.com/shriramsingh/
|
|
765
|
-
- **License**:
|
|
762
|
+
- **Community Repository**: [github.com/shriramsingh/prompttest-community](https://github.com/shriramsingh/prompttest-community)
|
|
763
|
+
- **Issues & Requests**: [File an Issue](https://github.com/shriramsingh/prompttest-community/issues)
|
|
764
|
+
- **License**: Business Source License 1.1 (BSL 1.1)
|