@aarwitz/tapp 0.17.19 → 0.17.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +2 -2
- package/Harness/OCQAHarnessUITests/ExplorerTests.swift +32 -9
- package/bin/tapp.js +2 -19
- package/mcp-server/src/index.js +31 -3
- package/mcp-server/src/project-config.js +18 -0
- package/package.json +1 -1
- package/scripts/flow_lib.py +26 -7
- package/scripts/quick-capture.sh +53 -31
- package/scripts/run-flow.sh +17 -11
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "tapp",
|
|
3
3
|
"description": "Give Claude hands and eyes on iOS, Android, and web apps, with exploration, replayable flows, evidence, and deterministic CI gates.",
|
|
4
|
-
"version": "0.17.
|
|
4
|
+
"version": "0.17.20",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Aaron Horowitz",
|
|
7
7
|
"url": "https://github.com/aarwitz"
|
|
@@ -24,7 +24,7 @@
|
|
|
24
24
|
"command": "npx",
|
|
25
25
|
"args": [
|
|
26
26
|
"-y",
|
|
27
|
-
"@aarwitz/tapp@0.17.
|
|
27
|
+
"@aarwitz/tapp@0.17.20",
|
|
28
28
|
"mcp"
|
|
29
29
|
],
|
|
30
30
|
"cwd": "${CLAUDE_PROJECT_DIR}"
|
|
@@ -715,8 +715,11 @@ class ExplorerTests: XCTestCase {
|
|
|
715
715
|
guard let emailF = emailField else { return ("no_login_form", "no email/username field visible") }
|
|
716
716
|
guard let passF = passwordField else { return ("no_login_form", "no password field visible") }
|
|
717
717
|
|
|
718
|
-
replaceText(on: emailF, with: email)
|
|
719
|
-
|
|
718
|
+
guard replaceText(on: emailF, with: email) else { return ("input_failed", lastTypeFailure) }
|
|
719
|
+
guard (emailF.value as? String) == email else {
|
|
720
|
+
return ("input_failed", "Email field did not retain the requested value; sign-in was not submitted")
|
|
721
|
+
}
|
|
722
|
+
guard replaceText(on: passF, with: password) else { return ("input_failed", lastTypeFailure) }
|
|
720
723
|
|
|
721
724
|
func formGone(within seconds: TimeInterval) -> Bool {
|
|
722
725
|
let deadline = Date().addingTimeInterval(seconds)
|
|
@@ -760,7 +763,7 @@ class ExplorerTests: XCTestCase {
|
|
|
760
763
|
// behavior). Detect and re-fill before submitting.
|
|
761
764
|
let pv = (passF.value as? String) ?? ""
|
|
762
765
|
if pv.isEmpty || pv == (passF.placeholderValue ?? "§none§") {
|
|
763
|
-
replaceText(on: passF, with: password)
|
|
766
|
+
guard replaceText(on: passF, with: password) else { return ("input_failed", lastTypeFailure) }
|
|
764
767
|
dismissKeyboardIfPresent()
|
|
765
768
|
}
|
|
766
769
|
submit = submitButton() ?? submit
|
|
@@ -807,7 +810,9 @@ class ExplorerTests: XCTestCase {
|
|
|
807
810
|
/// titles we have actually been on; a leading navigation-bar button named after one of them
|
|
808
811
|
/// is a genuine back control, while a custom leading action (Share, Delete) is not.
|
|
809
812
|
private func noteNavigationTitle() {
|
|
810
|
-
let
|
|
813
|
+
let bar = app.navigationBars.firstMatch
|
|
814
|
+
guard bar.exists else { return }
|
|
815
|
+
let title = bar.identifier
|
|
811
816
|
guard !title.isEmpty, backTitleHistory.last != title else { return }
|
|
812
817
|
backTitleHistory.append(title)
|
|
813
818
|
if backTitleHistory.count > 20 { backTitleHistory.removeFirst() }
|
|
@@ -5090,6 +5095,7 @@ class ExplorerTests: XCTestCase {
|
|
|
5090
5095
|
/// a tap nudged to the top of the frame. Three attempts, ~4 s worst case.
|
|
5091
5096
|
private func focusForTyping(_ element: XCUIElement) -> Bool {
|
|
5092
5097
|
func focused() -> Bool { (element.value(forKey: "hasKeyboardFocus") as? Bool) ?? false }
|
|
5098
|
+
if focused() { return true }
|
|
5093
5099
|
func settleUntilFocused() -> Bool {
|
|
5094
5100
|
let deadline = Date().addingTimeInterval(1.2)
|
|
5095
5101
|
repeat {
|
|
@@ -5157,11 +5163,28 @@ class ExplorerTests: XCTestCase {
|
|
|
5157
5163
|
return false
|
|
5158
5164
|
}
|
|
5159
5165
|
|
|
5160
|
-
|
|
5161
|
-
|
|
5162
|
-
|
|
5163
|
-
|
|
5164
|
-
|
|
5166
|
+
func hasContent() -> Bool {
|
|
5167
|
+
guard let value = element.value as? String else { return false }
|
|
5168
|
+
return !value.isEmpty && value != element.placeholderValue
|
|
5169
|
+
}
|
|
5170
|
+
if hasContent() {
|
|
5171
|
+
// A tap may put the cursor in the middle. Select the complete field instead of
|
|
5172
|
+
// deleting only the prefix, then prove it is empty before inserting new text.
|
|
5173
|
+
element.typeKey("a", modifierFlags: .command)
|
|
5174
|
+
element.typeText(XCUIKeyboardKey.delete.rawValue)
|
|
5175
|
+
if hasContent(), let remaining = element.value as? String {
|
|
5176
|
+
// Some simulator keyboard configurations ignore Command-A. Clear on both
|
|
5177
|
+
// sides of the cursor in that case; backspace alone leaves the suffix intact.
|
|
5178
|
+
let count = remaining.utf16.count
|
|
5179
|
+
element.typeText(String(repeating: XCUIKeyboardKey.delete.rawValue, count: count)
|
|
5180
|
+
+ String(repeating: XCUIKeyboardKey.forwardDelete.rawValue, count: count))
|
|
5181
|
+
}
|
|
5182
|
+
let deadline = Date().addingTimeInterval(1.0)
|
|
5183
|
+
while hasContent() && Date() < deadline { Thread.sleep(forTimeInterval: 0.1) }
|
|
5184
|
+
guard !hasContent() else {
|
|
5185
|
+
lastTypeFailure = "Field contents could not be cleared; replacement was not entered"
|
|
5186
|
+
return false
|
|
5187
|
+
}
|
|
5165
5188
|
}
|
|
5166
5189
|
|
|
5167
5190
|
element.typeText(text)
|
package/bin/tapp.js
CHANGED
|
@@ -102,13 +102,8 @@ function bootBestSimulator(preferredName = "iPhone 16 Pro") {
|
|
|
102
102
|
}
|
|
103
103
|
|
|
104
104
|
function harnessXctestrun() {
|
|
105
|
-
const
|
|
106
|
-
|
|
107
|
-
const found = fs.readdirSync(dir).find((f) => f.endsWith(".xctestrun"));
|
|
108
|
-
return found ? path.join(dir, found) : null;
|
|
109
|
-
} catch {
|
|
110
|
-
return null;
|
|
111
|
-
}
|
|
105
|
+
const result = run("bash", [path.join(packageRoot, "scripts/quick-capture.sh"), "harness-path"]);
|
|
106
|
+
return result.code === 0 && result.stdout ? result.stdout : null;
|
|
112
107
|
}
|
|
113
108
|
|
|
114
109
|
// Flags/positionals for the zero-config verbs (qa/open/tree/shot). `--key value` or bare `--key`.
|
|
@@ -172,17 +167,6 @@ function requireMacFor(what) {
|
|
|
172
167
|
process.exit(1);
|
|
173
168
|
}
|
|
174
169
|
|
|
175
|
-
function ensureIOSHarness() {
|
|
176
|
-
const result = spawnSync("bash", [path.join(packageRoot, "scripts", "quick-capture.sh"), "build-harness"], {
|
|
177
|
-
stdio: "inherit",
|
|
178
|
-
env: process.env,
|
|
179
|
-
});
|
|
180
|
-
if ((result.status ?? 1) !== 0) {
|
|
181
|
-
console.error("❌ Could not prepare the iOS test harness.");
|
|
182
|
-
process.exit(result.status ?? 1);
|
|
183
|
-
}
|
|
184
|
-
}
|
|
185
|
-
|
|
186
170
|
function requestedPlatform(flags, target = "") {
|
|
187
171
|
if (typeof flags.platform === "string") return flags.platform.toLowerCase();
|
|
188
172
|
if (/^https?:\/\//i.test(target)) return "web";
|
|
@@ -1291,7 +1275,6 @@ switch (command) {
|
|
|
1291
1275
|
invocation = [process.execPath, [path.join(packageRoot, "scripts", "run-android-flow.js"), absolute, target.appId, target.apkPath || "", target.serial || ""]];
|
|
1292
1276
|
} else {
|
|
1293
1277
|
requireMacFor("iOS Flow replay");
|
|
1294
|
-
ensureIOSHarness();
|
|
1295
1278
|
invocation = ["bash", [path.join(packageRoot, "scripts", "run-flow.sh"), absolute, typeof flags["bundle-id"] === "string" ? flags["bundle-id"] : flow.app || ""]];
|
|
1296
1279
|
}
|
|
1297
1280
|
const result = spawnSync(invocation[0], invocation[1], { stdio: "inherit", env });
|
package/mcp-server/src/index.js
CHANGED
|
@@ -649,6 +649,18 @@ export async function resolveAppTarget(input, { cwd = process.cwd(), onStatus =
|
|
|
649
649
|
let activeSession = null;
|
|
650
650
|
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
651
651
|
|
|
652
|
+
export function sessionFailureReason(log = "", credentials = {}) {
|
|
653
|
+
for (const pattern of [/Failed to get matching snapshot:[^\n]*/, /Failed to synthesize event:[^\n]*/, /error:\s*[^\n]+/, /Error Domain=[^\n]+/, /Test Case '[^']*' failed[^\n]*/]) {
|
|
654
|
+
const match = String(log).match(pattern);
|
|
655
|
+
if (match) {
|
|
656
|
+
let message = match[0].trim();
|
|
657
|
+
for (const secret of Object.values(credentials).filter((value) => typeof value === "string" && value)) message = message.split(secret).join("[redacted]");
|
|
658
|
+
return message.slice(0, 1200);
|
|
659
|
+
}
|
|
660
|
+
}
|
|
661
|
+
return "The XCTest session process exited; reopen the app to start a new session.";
|
|
662
|
+
}
|
|
663
|
+
|
|
652
664
|
function consumeSessionStdout(chunk) {
|
|
653
665
|
if (!activeSession) return;
|
|
654
666
|
activeSession.buffer += chunk;
|
|
@@ -765,9 +777,9 @@ async function startSession(bundleId, extraEnv = {}) {
|
|
|
765
777
|
while (!activeSession.ready && Date.now() < deadline && !activeSession.ended) await sleep(300);
|
|
766
778
|
if (activeSession.ended) {
|
|
767
779
|
// Surface the real failure from the harness output instead of a shrug.
|
|
768
|
-
const errLine = (
|
|
780
|
+
const errLine = sessionFailureReason(activeSession.buffer, activeSession.creds);
|
|
769
781
|
activeSession = null;
|
|
770
|
-
return { error: `Session process exited before it became ready
|
|
782
|
+
return { error: `Session process exited before it became ready — ${errLine}` };
|
|
771
783
|
}
|
|
772
784
|
if (!activeSession.ready) { return { error: "Session did not become ready within the time limit." }; }
|
|
773
785
|
|
|
@@ -1014,9 +1026,14 @@ export function sessionActUsageError(cmd = {}) {
|
|
|
1014
1026
|
async function sessionAct(cmd) {
|
|
1015
1027
|
const startedAt = Date.now();
|
|
1016
1028
|
const done = (result) => ({ ...result, durationMs: Date.now() - startedAt });
|
|
1017
|
-
if (!activeSession
|
|
1029
|
+
if (!activeSession) return done({ error: "No active session. Call tapp_session_start first." });
|
|
1030
|
+
if (activeSession.ended) return done({ error: sessionFailureReason(activeSession.buffer, activeSession.creds) });
|
|
1018
1031
|
const usage = sessionActUsageError(cmd);
|
|
1019
1032
|
if (usage) return done({ status: "usage", detail: usage, ...treeSnapshot(), recordedSteps: activeSession.recording.length });
|
|
1033
|
+
if (cmd.action === "login") {
|
|
1034
|
+
if (cmd.email) activeSession.creds.email = cmd.email;
|
|
1035
|
+
if (cmd.password) activeSession.creds.password = cmd.password;
|
|
1036
|
+
}
|
|
1020
1037
|
let coordinateResolvedTarget = "";
|
|
1021
1038
|
if (cmd.action === "tap" && !cmd.id && Number.isFinite(cmd.x) && Number.isFinite(cmd.y)) {
|
|
1022
1039
|
coordinateResolvedTarget = semanticTargetAtPoint(activeSession.latestTree?.elements, cmd.x, cmd.y);
|
|
@@ -1176,6 +1193,10 @@ async function sessionAct(cmd) {
|
|
|
1176
1193
|
const td = Date.now() + 5_000;
|
|
1177
1194
|
while (activeSession.treeVersion === beforeVer && Date.now() < td && !activeSession.ended) await sleep(150);
|
|
1178
1195
|
const snap = treeSnapshot();
|
|
1196
|
+
if (status === "timeout" && activeSession.ended) {
|
|
1197
|
+
status = "harness_failed";
|
|
1198
|
+
detail = sessionFailureReason(activeSession.buffer, activeSession.creds);
|
|
1199
|
+
}
|
|
1179
1200
|
if (status === "ok") recordStep(cmd, snap); // record only successful acts
|
|
1180
1201
|
return done({ status, typedInto, detail, ...snap, recordedSteps: activeSession ? activeSession.recording.length : 0, ...(coordinateResolvedTarget ? { coordinateResolvedTarget } : {}) });
|
|
1181
1202
|
}
|
|
@@ -3272,6 +3293,7 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
3272
3293
|
action: { type: "string", enum: ["login", "tap", "type", "swipe", "back", "wait", "tree", "screenshot"] },
|
|
3273
3294
|
email: { type: "string", description: "login: email/username to sign in with" },
|
|
3274
3295
|
password: { type: "string", description: "login: password to sign in with" },
|
|
3296
|
+
actor: { type: "string", description: "login: configured actor; explicit email/password override its environment bindings" },
|
|
3275
3297
|
id: { type: "string", description: "Element accessibility id or visible/partial label (for tap/type/wait)" },
|
|
3276
3298
|
x: { type: "number", description: "Tap X coordinate (points), if not using id" },
|
|
3277
3299
|
y: { type: "number", description: "Tap Y coordinate (points), if not using id" },
|
|
@@ -4585,6 +4607,12 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
4585
4607
|
if (action === "login") {
|
|
4586
4608
|
if (isNonEmptyString(args.email)) cmd.email = args.email.trim();
|
|
4587
4609
|
if (isNonEmptyString(args.password)) cmd.password = args.password;
|
|
4610
|
+
if (isNonEmptyString(args.actor)) {
|
|
4611
|
+
try {
|
|
4612
|
+
const { resolveActorCredentials } = await import("./project-config.js");
|
|
4613
|
+
Object.assign(cmd, resolveActorCredentials(workspaceRoot, { actor: args.actor.trim(), email: cmd.email, password: cmd.password }));
|
|
4614
|
+
} catch (error) { return errorResult(error.message); }
|
|
4615
|
+
}
|
|
4588
4616
|
}
|
|
4589
4617
|
if (action === "wait") cmd.timeoutMs = Math.max(500, Math.min(60_000, asInteger(args.timeoutMs, 5000)));
|
|
4590
4618
|
const r = await sessionAct(cmd);
|
|
@@ -4,6 +4,24 @@ import { TAPP_DIRECTORY, projectArtifactDirectory } from "./project-paths.js";
|
|
|
4
4
|
|
|
5
5
|
export const PROJECT_CONFIG_RELATIVE_PATH = `${TAPP_DIRECTORY}/project.json`;
|
|
6
6
|
|
|
7
|
+
// Resolve a named actor entirely inside the engine. Never return these values as tool
|
|
8
|
+
// output or fall back to another identity when a binding is absent.
|
|
9
|
+
export function resolveActorCredentials(projectDir, { actor, email, password, env = process.env } = {}) {
|
|
10
|
+
const loaded = readProjectConfig(projectDir);
|
|
11
|
+
if (loaded.errors.length) throw new Error(`Invalid ${loaded.relativePath}: ${loaded.errors.join("; ")}`);
|
|
12
|
+
const configured = loaded.config.actors?.[actor];
|
|
13
|
+
if (!configured) throw new Error(`Actor '${actor}' is not configured in ${loaded.relativePath}`);
|
|
14
|
+
const credentials = { email, password };
|
|
15
|
+
for (const key of ["email", "password"]) {
|
|
16
|
+
if (typeof credentials[key] === "string" && credentials[key]) continue;
|
|
17
|
+
const binding = configured.credentials?.[key]?.env;
|
|
18
|
+
if (!binding) throw new Error(`Actor '${actor}' has no ${key} environment binding`);
|
|
19
|
+
if (!env[binding]) throw new Error(`Actor '${actor}' requires $${binding} in the Tapp process environment`);
|
|
20
|
+
credentials[key] = env[binding];
|
|
21
|
+
}
|
|
22
|
+
return credentials;
|
|
23
|
+
}
|
|
24
|
+
|
|
7
25
|
const ACTOR_NAME = /^[A-Za-z][A-Za-z0-9_-]{0,63}$/;
|
|
8
26
|
const ENV_NAME = /^[A-Z_][A-Z0-9_]{0,127}$/;
|
|
9
27
|
const CREDENTIAL_NAME = /^[a-z][A-Za-z0-9_-]{0,63}$/;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@aarwitz/tapp",
|
|
3
|
-
"version": "0.17.
|
|
3
|
+
"version": "0.17.20",
|
|
4
4
|
"mcpName": "io.github.aarwitz/tapp",
|
|
5
5
|
"description": "Let coding agents verify UI changes on real iOS, Android, and web surfaces, then enforce reviewed proof in deterministic CI.",
|
|
6
6
|
"license": "MIT",
|
package/scripts/flow_lib.py
CHANGED
|
@@ -10,6 +10,7 @@ Usage:
|
|
|
10
10
|
flow_lib.py report --json <harness.log> # prints machine JSON {passed,total,failed,steps}
|
|
11
11
|
"""
|
|
12
12
|
import json
|
|
13
|
+
import os
|
|
13
14
|
import re
|
|
14
15
|
import sys
|
|
15
16
|
|
|
@@ -73,23 +74,34 @@ def raw_json(path):
|
|
|
73
74
|
ABORT_PATTERNS = [
|
|
74
75
|
r"Failed to synthesize event: [^\n]*",
|
|
75
76
|
r"Neither element nor any descendant has keyboard focus[^\n]*",
|
|
76
|
-
r"Test Case '[^']*' failed \([^)]*\)",
|
|
77
77
|
r"Error Domain=[^\n]*",
|
|
78
78
|
r"Unable to (?:find|launch|boot)[^\n]*",
|
|
79
79
|
r"App state is [^\n]*",
|
|
80
80
|
r"Timed out [^\n]*",
|
|
81
81
|
r"Testing failed:[^\n]*",
|
|
82
82
|
r"error: [^\n]*",
|
|
83
|
+
r"Test Case '[^']*' failed \([^)]*\)",
|
|
83
84
|
r"\*\* TEST (?:EXECUTE )?FAILED \*\*",
|
|
84
85
|
]
|
|
85
86
|
|
|
86
87
|
|
|
87
|
-
def harness_abort_reason(log):
|
|
88
|
+
def harness_abort_reason(log, path=None):
|
|
88
89
|
"""First XCTest/xcodebuild line that explains why a run ended before step 1."""
|
|
90
|
+
if path:
|
|
91
|
+
for summary_path in [path + ".xctest-summary.json", os.path.join(os.path.dirname(path), "xctest-summary.json")]:
|
|
92
|
+
try:
|
|
93
|
+
with open(summary_path, encoding="utf-8") as summary_file:
|
|
94
|
+
summary = json.load(summary_file)
|
|
95
|
+
messages = [failure["failureText"] for failure in summary.get("testFailures", [])
|
|
96
|
+
if isinstance(failure.get("failureText"), str) and failure["failureText"].strip()]
|
|
97
|
+
if messages:
|
|
98
|
+
return " | ".join(dict.fromkeys(messages))[:1200]
|
|
99
|
+
except (OSError, ValueError, TypeError, AttributeError):
|
|
100
|
+
pass
|
|
89
101
|
for pat in ABORT_PATTERNS:
|
|
90
102
|
m = re.search(pat, log)
|
|
91
103
|
if m:
|
|
92
|
-
return m.group(0).strip()[:
|
|
104
|
+
return m.group(0).strip()[:1200]
|
|
93
105
|
return None
|
|
94
106
|
|
|
95
107
|
|
|
@@ -102,7 +114,14 @@ def report(path, as_json=False):
|
|
|
102
114
|
declared_total = None
|
|
103
115
|
for line in log.splitlines():
|
|
104
116
|
line = line.strip()
|
|
105
|
-
if line.startswith("
|
|
117
|
+
if line.startswith("OCQA_FLOW_PLAN:"):
|
|
118
|
+
try:
|
|
119
|
+
plan = json.loads(line[len("OCQA_FLOW_PLAN:"):])
|
|
120
|
+
name = plan.get("name", name)
|
|
121
|
+
declared_total = plan.get("total", declared_total)
|
|
122
|
+
except (ValueError, TypeError, AttributeError):
|
|
123
|
+
pass
|
|
124
|
+
elif line.startswith("OCQA_FLOW_STEP:"):
|
|
106
125
|
try:
|
|
107
126
|
steps.append(json.loads(line[len("OCQA_FLOW_STEP:"):]))
|
|
108
127
|
except Exception:
|
|
@@ -124,7 +143,7 @@ def report(path, as_json=False):
|
|
|
124
143
|
except Exception:
|
|
125
144
|
pass
|
|
126
145
|
|
|
127
|
-
total = (result or {}).get("total", len(steps))
|
|
146
|
+
total = (result or {}).get("total", declared_total if declared_total is not None else len(steps))
|
|
128
147
|
failed = (result or {}).get("failed", sum(1 for s in steps if s.get("status") == "fail"))
|
|
129
148
|
passed = (result or {}).get("passed", failed == 0 and bool(steps))
|
|
130
149
|
|
|
@@ -136,7 +155,7 @@ def report(path, as_json=False):
|
|
|
136
155
|
# failure of the first step so the scoreboard says why instead of looking like a crash.
|
|
137
156
|
abort_reason = None
|
|
138
157
|
if not steps and not passed:
|
|
139
|
-
abort_reason = harness_abort_reason(log)
|
|
158
|
+
abort_reason = harness_abort_reason(log, path)
|
|
140
159
|
steps.append({"index": 1, "action": "harness", "target": "", "status": "fail",
|
|
141
160
|
"detail": abort_reason or "the harness exited before the first step; see flow.log"})
|
|
142
161
|
failed = max(failed, 1)
|
|
@@ -144,7 +163,7 @@ def report(path, as_json=False):
|
|
|
144
163
|
# The harness died MID-run (an XCTest assertion, the test time budget, a crash): there is
|
|
145
164
|
# no final OCQA_FLOW_RESULT line. This used to be reported as "PASSED 16/16" because
|
|
146
165
|
# `total` silently became the number of steps that happened to run before the death.
|
|
147
|
-
abort_reason = harness_abort_reason(log)
|
|
166
|
+
abort_reason = harness_abort_reason(log, path)
|
|
148
167
|
total = declared_total or total
|
|
149
168
|
steps.append({"index": len(steps) + 1, "action": "harness", "target": "", "status": "fail",
|
|
150
169
|
"detail": (abort_reason or "the harness exited") + f" — run ended after step {executed} of {total}"})
|
package/scripts/quick-capture.sh
CHANGED
|
@@ -67,44 +67,58 @@ cleanup_stale_recorders() {
|
|
|
67
67
|
sleep 1
|
|
68
68
|
}
|
|
69
69
|
|
|
70
|
+
harness_fingerprint() {
|
|
71
|
+
# SDK names in .xctestrun filenames describe the BUILD SDK, not the simulator OS.
|
|
72
|
+
# Record the source/project and selected toolchain, then reuse the exact build product.
|
|
73
|
+
{
|
|
74
|
+
find "$(dirname "$HARNESS_PROJECT")" -type f \( -name '*.swift' -o -name project.pbxproj -o -name '*.xcscheme' -o -name '*.plist' \) -print0 \
|
|
75
|
+
| sort -z | xargs -0 shasum
|
|
76
|
+
xcode-select -p
|
|
77
|
+
xcodebuild -version
|
|
78
|
+
xcrun --sdk iphonesimulator --show-sdk-version
|
|
79
|
+
} | shasum | cut -c1-40
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
cached_harness_path() {
|
|
83
|
+
local marker="$HARNESS_DERIVED/.last-sim-udid"
|
|
84
|
+
[[ -f "$marker" ]] || return 1
|
|
85
|
+
[[ "$(sed -n 1p "$marker")" == "$UDID" ]] || return 1
|
|
86
|
+
[[ "$(sed -n 2p "$marker")" == "$(harness_fingerprint)" ]] || return 1
|
|
87
|
+
local selected="$(sed -n 3p "$marker")"
|
|
88
|
+
[[ -n "$selected" && "$selected" != */* && "$selected" == *.xctestrun ]] || return 1
|
|
89
|
+
[[ -f "$HARNESS_DERIVED/Build/Products/$selected" ]] || return 1
|
|
90
|
+
printf '%s\n' "$HARNESS_DERIVED/Build/Products/$selected"
|
|
91
|
+
}
|
|
92
|
+
|
|
70
93
|
ensure_harness_built() {
|
|
71
94
|
local sim_name="${1:-iPhone 16 Pro}"
|
|
72
95
|
local udid="${UDID:-}"
|
|
73
96
|
local marker="$HARNESS_DERIVED/.last-sim-udid"
|
|
74
|
-
local
|
|
75
|
-
if
|
|
76
|
-
last_udid=$(sed -n 1p "$marker" 2>/dev/null || true)
|
|
77
|
-
last_fp=$(sed -n 2p "$marker" 2>/dev/null || true)
|
|
78
|
-
fi
|
|
79
|
-
# Fingerprint the harness source so a package update actually reaches users — without this,
|
|
80
|
-
# a warm cache serves the OLD harness forever (npm normalizes mtimes, so hash the content).
|
|
81
|
-
local src_file="$(dirname "$HARNESS_PROJECT")/OCQAHarnessUITests/ExplorerTests.swift"
|
|
82
|
-
local src_fp=$(shasum "$src_file" 2>/dev/null | cut -c1-12)
|
|
83
|
-
local xctestrun=$(find "$HARNESS_DERIVED/Build/Products" -name "*.xctestrun" 2>/dev/null | head -1)
|
|
84
|
-
|
|
85
|
-
# Reuse the cached harness ONLY if it was built for the currently-booted simulator AND from
|
|
86
|
-
# the same harness sources. A harness built for a different sim can fail to launch the
|
|
87
|
-
# interactive session; a stale-source harness silently lacks shipped fixes.
|
|
88
|
-
if [[ -n "$xctestrun" && ( -z "$udid" || "$udid" == "$last_udid" ) && "$src_fp" == "$last_fp" ]]; then
|
|
97
|
+
local xctestrun
|
|
98
|
+
if xctestrun=$(cached_harness_path); then
|
|
89
99
|
echo "Harness already built for this sim: $xctestrun" >&2
|
|
90
100
|
return 0
|
|
91
101
|
fi
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
102
|
+
echo "Preparing harness for $sim_name (source, simulator, or Xcode cache changed)..." >&2
|
|
103
|
+
local src_fp=$(harness_fingerprint)
|
|
104
|
+
mkdir -p "$HARNESS_DERIVED/Build/Products"
|
|
105
|
+
rm -f "$marker"
|
|
106
|
+
# Xcode leaves old SDK descriptors beside the new one. Only the descriptor emitted by
|
|
107
|
+
# this successful build may become current; never select an arbitrary directory entry.
|
|
108
|
+
find "$HARNESS_DERIVED/Build/Products" -maxdepth 1 -name '*.xctestrun' -type f -delete
|
|
99
109
|
xcodebuild build-for-testing \
|
|
100
110
|
-project "$HARNESS_PROJECT" \
|
|
101
111
|
-scheme OCQAHarnessUITests \
|
|
102
112
|
-destination "platform=iOS Simulator,id=$UDID" \
|
|
103
113
|
-derivedDataPath "$HARNESS_DERIVED" \
|
|
104
114
|
2>&1 | tail -5 >&2
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
115
|
+
local products=()
|
|
116
|
+
while IFS= read -r product; do products+=("$product"); done < <(find "$HARNESS_DERIVED/Build/Products" -maxdepth 1 -name '*.xctestrun' -type f)
|
|
117
|
+
if [[ ${#products[@]} -ne 1 ]]; then
|
|
118
|
+
echo "ERROR: Harness build produced ${#products[@]} test descriptors; no cache was selected." >&2
|
|
119
|
+
return 1
|
|
120
|
+
fi
|
|
121
|
+
printf '%s\n%s\n%s\n' "$udid" "$src_fp" "$(basename "${products[0]}")" > "$marker"
|
|
108
122
|
}
|
|
109
123
|
|
|
110
124
|
run_harness_test() {
|
|
@@ -189,7 +203,7 @@ run_harness_test() {
|
|
|
189
203
|
}
|
|
190
204
|
CONF
|
|
191
205
|
|
|
192
|
-
local xctestrun=$(
|
|
206
|
+
local xctestrun=$(cached_harness_path)
|
|
193
207
|
if [[ -z "$xctestrun" ]]; then
|
|
194
208
|
echo "ERROR: No xctestrun found. Run: tapp install" >&2
|
|
195
209
|
return 1
|
|
@@ -226,15 +240,23 @@ while [[ $# -gt 0 ]]; do
|
|
|
226
240
|
esac
|
|
227
241
|
done
|
|
228
242
|
|
|
229
|
-
mkdir -p "$CAPTURE_DIR"
|
|
230
243
|
UDID=$(get_booted_sim)
|
|
231
244
|
SIM_NAME=$(get_sim_name)
|
|
232
245
|
|
|
233
246
|
if [[ -z "$UDID" ]]; then
|
|
234
|
-
echo "ERROR: No booted simulator found. Boot one first:"
|
|
235
|
-
echo " xcrun simctl boot 'iPhone 16 Pro'"
|
|
247
|
+
echo "ERROR: No booted simulator found. Boot one first:" >&2
|
|
248
|
+
echo " xcrun simctl boot 'iPhone 16 Pro'" >&2
|
|
236
249
|
exit 1
|
|
237
250
|
fi
|
|
251
|
+
if [[ "$MODE" == "harness-path" ]]; then
|
|
252
|
+
cached_harness_path
|
|
253
|
+
exit $?
|
|
254
|
+
elif [[ "$MODE" == "prepare-harness" ]]; then
|
|
255
|
+
ensure_harness_built "$SIM_NAME"
|
|
256
|
+
cached_harness_path
|
|
257
|
+
exit $?
|
|
258
|
+
fi
|
|
259
|
+
mkdir -p "$CAPTURE_DIR"
|
|
238
260
|
echo "Simulator: $SIM_NAME ($UDID)"
|
|
239
261
|
echo "Output: $CAPTURE_DIR"
|
|
240
262
|
echo ""
|
|
@@ -452,7 +474,7 @@ OCQA_COMPLETE:{\"actions\":0,\"states\":0,\"issues\":1,\"screens\":\"\",\"outcom
|
|
|
452
474
|
"OCQA_TEST_PASSWORD": "${OCQA_TEST_PASSWORD:-}"$sess_args_line$sess_env_line
|
|
453
475
|
}
|
|
454
476
|
CONF
|
|
455
|
-
xctestrun=$(
|
|
477
|
+
xctestrun=$(cached_harness_path)
|
|
456
478
|
if [[ -z "$xctestrun" ]]; then
|
|
457
479
|
echo "ERROR: No xctestrun. Run: tapp install" >&2
|
|
458
480
|
exit 1
|
|
@@ -470,7 +492,7 @@ CONF
|
|
|
470
492
|
# for a fast first tool call later. No capture output.
|
|
471
493
|
ensure_harness_built "$SIM_NAME"
|
|
472
494
|
rmdir "$CAPTURE_DIR" 2>/dev/null || true
|
|
473
|
-
echo "Harness ready: $(
|
|
495
|
+
echo "Harness ready: $(cached_harness_path)"
|
|
474
496
|
;;
|
|
475
497
|
|
|
476
498
|
*)
|
package/scripts/run-flow.sh
CHANGED
|
@@ -24,19 +24,13 @@ APP="${2:-}"
|
|
|
24
24
|
|
|
25
25
|
UDID="$(xcrun simctl list devices booted -j 2>/dev/null | python3 -c 'import sys,json; d=json.load(sys.stdin); print(next((x["udid"] for v in d["devices"].values() for x in v if x.get("state")=="Booted"), ""))')"
|
|
26
26
|
[ -z "$UDID" ] && { echo "❌ No booted simulator."; exit 2; }
|
|
27
|
-
#
|
|
28
|
-
XCTR=""
|
|
29
|
-
TAPP_RUNTIME_HOME="${TAPP_HOME:-}"
|
|
30
|
-
[ -n "$TAPP_RUNTIME_HOME" ] && XCTR="$(find "$TAPP_RUNTIME_HOME/harness-derived/Build/Products" -name '*.xctestrun' 2>/dev/null | head -1)"
|
|
31
|
-
[ -z "$XCTR" ] && XCTR="$(find "$HOME/Library/Developer/Xcode/DerivedData/OCQAHarness-"*/Build/Products -name '*.xctestrun' 2>/dev/null | head -1)"
|
|
32
|
-
[ -z "$XCTR" ] && XCTR="$(find /tmp/tapp-harness-derived/Build/Products -name '*.xctestrun' 2>/dev/null | head -1)"
|
|
33
|
-
[ -z "$XCTR" ] && XCTR="$(find /tmp/harness-build/Build/Products -name '*.xctestrun' 2>/dev/null | head -1)"
|
|
34
|
-
[ -z "$XCTR" ] && { echo "❌ Harness not built. Run: tapp install"; exit 2; }
|
|
27
|
+
# CLI, MCP, sessions, and doctor share the same source/toolchain-validated descriptor.
|
|
28
|
+
XCTR="$(bash "$ROOT/scripts/quick-capture.sh" prepare-harness)" || exit 2
|
|
35
29
|
|
|
36
30
|
NAME="$(python3 -c "import sys,json;print(json.loads(sys.argv[1]).get('name','flow'))" "$FLOW_JSON")"
|
|
37
31
|
echo "▶️ Running flow \"$NAME\" against $APP …"
|
|
38
32
|
|
|
39
|
-
TOKEN="
|
|
33
|
+
TOKEN="$$-$(date +%s)"
|
|
40
34
|
CFG="/tmp/ocqa-flow-$TOKEN.json"
|
|
41
35
|
AI_RESP="/tmp/ocqa-flow-ai-$TOKEN.json"
|
|
42
36
|
AI_DIR="/tmp/ocqa-flow-ai-$TOKEN"
|
|
@@ -83,10 +77,15 @@ if [ "$?" -ne 0 ]; then
|
|
|
83
77
|
fi
|
|
84
78
|
|
|
85
79
|
LOG="${FLOW_LOG:-/tmp/ocqa-flow-$TOKEN.log}"
|
|
80
|
+
python3 - "$FLOW_JSON" > "$LOG" <<'PY'
|
|
81
|
+
import json, sys
|
|
82
|
+
flow = json.loads(sys.argv[1])
|
|
83
|
+
print("OCQA_FLOW_PLAN:" + json.dumps({"name": flow.get("name", "flow"), "total": len(flow.get("steps", []))}))
|
|
84
|
+
PY
|
|
86
85
|
# assert_ai judge sidecar (only when a key is present) — same file-channel as vision escalation.
|
|
87
86
|
RESPONDER_PID=""
|
|
88
87
|
if [ -n "${ANTHROPIC_API_KEY:-}" ]; then
|
|
89
|
-
mkdir -p "$AI_DIR"
|
|
88
|
+
mkdir -p "$AI_DIR"
|
|
90
89
|
python3 "$ROOT/scripts/flow_ai_judge.py" "$LOG" "$AI_RESP" > "/tmp/ocqa-flow-judge-$TOKEN.log" 2>&1 &
|
|
91
90
|
RESPONDER_PID=$!
|
|
92
91
|
fi
|
|
@@ -94,10 +93,17 @@ fi
|
|
|
94
93
|
TEST_RUNNER_OCQA_CONFIG_PATH="$CFG" xcodebuild test-without-building \
|
|
95
94
|
-xctestrun "$XCTR" -destination "platform=iOS Simulator,id=$UDID" \
|
|
96
95
|
-only-testing:"OCQAHarnessUITests/ExplorerTests/testReplayFlow" \
|
|
97
|
-
-resultBundlePath "$RESULT_BUNDLE"
|
|
96
|
+
-resultBundlePath "$RESULT_BUNDLE" >> "$LOG" 2>&1
|
|
97
|
+
XCODE_STATUS=$?
|
|
98
98
|
[ -n "$RESPONDER_PID" ] && { kill "$RESPONDER_PID" 2>/dev/null; wait "$RESPONDER_PID" 2>/dev/null; }
|
|
99
99
|
|
|
100
100
|
cp "$LOG" "$EVIDENCE_DIR/flow.log"
|
|
101
|
+
# XCTest can put the actionable failure only in the result bundle. Keep that summary
|
|
102
|
+
# beside the log so both the CLI and MCP report readers surface it before generic status.
|
|
103
|
+
if [ "$XCODE_STATUS" -ne 0 ] && [ -d "$RESULT_BUNDLE" ]; then
|
|
104
|
+
xcrun xcresulttool get test-results summary --path "$RESULT_BUNDLE" > "$EVIDENCE_DIR/xctest-summary.json" 2> "$EVIDENCE_DIR/xcresulttool.log" || true
|
|
105
|
+
cp "$EVIDENCE_DIR/xctest-summary.json" "$LOG.xctest-summary.json"
|
|
106
|
+
fi
|
|
101
107
|
grep '^OCQA_EVIDENCE_WARNING:' "$LOG" >&2 || true
|
|
102
108
|
python3 "$ROOT/scripts/flow_lib.py" report --json "$LOG" > "$EVIDENCE_DIR/flow-report.json"
|
|
103
109
|
echo ""
|