@mobileaidev/ai-app-bridge 0.3.7 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/README.md +52 -39
  2. package/bin/ai-app-bridge.js +56 -17
  3. package/bin/command-discovery.js +19 -4
  4. package/bin/command-registry.js +21 -6
  5. package/bin/command-request.js +68 -4
  6. package/bin/execution-host.js +32 -5
  7. package/bin/execution-runtime.js +17 -10
  8. package/bin/executors/command-schema.js +1 -1
  9. package/bin/executors/preparation.js +262 -0
  10. package/bin/extraction/json-value.js +26 -0
  11. package/bin/extraction/prepare.js +57 -0
  12. package/bin/extraction/regex.js +30 -0
  13. package/bin/extraction/runner.js +77 -0
  14. package/bin/intent/install-intent.js +3 -1
  15. package/bin/ios-provider.js +2 -2
  16. package/bin/ios-wda-project.js +48 -1
  17. package/bin/mcp-server.js +15 -20
  18. package/bin/public-reply.js +184 -0
  19. package/bin/response-store.js +60 -0
  20. package/bin/runtime-client.js +32 -15
  21. package/bin/runtime-directory.js +37 -8
  22. package/bin/script/node-runtime-adapter.js +139 -123
  23. package/bin/script/python-runtime-adapter.js +1 -1
  24. package/bin/script/script-diagnostics.js +21 -0
  25. package/bin/script/script-durable-restore.js +1 -0
  26. package/bin/script/script-host-port.js +3 -2
  27. package/bin/script/script-sdk.js +39 -4
  28. package/bin/script/script-sdk.py +79 -7
  29. package/bin/script/script-session-channel.js +27 -10
  30. package/bin/script/script-supervisor.js +8 -0
  31. package/bin/shared-kernel/argument-schema.js +44 -12
  32. package/bin/shared-kernel/evidence-archive.js +2 -2
  33. package/bin/shared-kernel/evidence-schema.js +16 -1
  34. package/bin/shared-kernel/evidence-store.js +3 -3
  35. package/bin/shared-kernel/execution-contracts.js +13 -5
  36. package/bin/shared-kernel/execution-target.js +4 -0
  37. package/bin/shared-kernel/request-context.js +2 -2
  38. package/bin/target-execution.js +2 -0
  39. package/docs/COMMAND_CONTRACT.md +91 -19
  40. package/docs/EVIDENCE_ARCHIVE.md +14 -1
  41. package/docs/INSTALLATION.md +73 -0
  42. package/docs/INTENT_FOREGROUND.md +4 -1
  43. package/docs/OPTIONAL_EXECUTORS.md +41 -13
  44. package/docs/RELEASE.md +60 -113
  45. package/docs/RESPONSE_EXTRACTION.md +122 -0
  46. package/docs/SCRIPT_AUTHORING.md +125 -6
  47. package/node_modules/@mobileaidev/segmented-fact-store-native/PREBUILDS.md +29 -0
  48. package/node_modules/@mobileaidev/segmented-fact-store-native/binding-path.js +29 -0
  49. package/node_modules/@mobileaidev/segmented-fact-store-native/binding.gyp +1 -0
  50. package/node_modules/@mobileaidev/segmented-fact-store-native/index.js +1 -3
  51. package/node_modules/@mobileaidev/segmented-fact-store-native/install.js +5 -0
  52. package/node_modules/@mobileaidev/segmented-fact-store-native/package.json +11 -5
  53. package/node_modules/@mobileaidev/segmented-fact-store-native/prebuilds/darwin-arm64/segmented_fact_store.node +0 -0
  54. package/node_modules/@mobileaidev/segmented-fact-store-native/prebuilds/darwin-x64/segmented_fact_store.node +0 -0
  55. package/node_modules/@mobileaidev/segmented-fact-store-native/prebuilds/linux-arm64-glibc/segmented_fact_store.node +0 -0
  56. package/node_modules/@mobileaidev/segmented-fact-store-native/prebuilds/linux-x64-glibc/segmented_fact_store.node +0 -0
  57. package/node_modules/@mobileaidev/segmented-fact-store-native/prebuilds/manifest.json +27 -0
  58. package/node_modules/@mobileaidev/segmented-fact-store-native/scripts/build-release-prebuilds.js +33 -0
  59. package/node_modules/@mobileaidev/segmented-fact-store-native/scripts/stage-prebuild.js +17 -0
  60. package/package.json +12 -5
  61. package/runtime/executors/android/prepare.init.gradle +92 -0
  62. package/runtime/executors/playwright/package-lock.json +2 -2
  63. package/runtime/executors/playwright/package.json +1 -1
  64. package/skills/ai-app-bridge-use/SKILL.md +19 -4
@@ -2,7 +2,8 @@
2
2
 
3
3
  const MAX_FRAME_BYTES = 1024 * 1024;
4
4
 
5
- function createScriptSessionChannel(child, { maxFrameBytes = MAX_FRAME_BYTES } = {}) {
5
+ function createScriptSessionChannel(child, { maxFrameBytes = MAX_FRAME_BYTES,
6
+ maxInputFrameBytes = maxFrameBytes, maxOutputFrameBytes = maxFrameBytes, maxDiagnosticBytes = 8192 } = {}) {
6
7
  let buffer = '';
7
8
  let closed = false;
8
9
  let failure = null;
@@ -12,10 +13,18 @@ function createScriptSessionChannel(child, { maxFrameBytes = MAX_FRAME_BYTES } =
12
13
  const pending = [];
13
14
  const replies = new Map();
14
15
  const maxPending = 32;
16
+ let stderr = Buffer.alloc(0);
17
+ let stderrTotalBytes = 0;
15
18
 
16
19
  if (child.stderr) {
17
- child.stderr.on('data', () => {});
20
+ child.stderr.on('data', chunk => {
21
+ const bytes = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
22
+ stderrTotalBytes += bytes.length;
23
+ stderr = bytes.length >= maxDiagnosticBytes ? Buffer.from(bytes.subarray(-maxDiagnosticBytes))
24
+ : Buffer.concat([stderr.subarray(Math.max(0, stderr.length + bytes.length - maxDiagnosticBytes)), bytes]);
25
+ });
18
26
  }
27
+ child.stdin.on('error', error => fail(error));
19
28
  child.stdout.setEncoding('utf8');
20
29
  child.stdout.on('data', (chunk) => {
21
30
  if (closed) return;
@@ -25,7 +34,7 @@ function createScriptSessionChannel(child, { maxFrameBytes = MAX_FRAME_BYTES } =
25
34
  const line = buffer.slice(0, index);
26
35
  buffer = buffer.slice(index + 1);
27
36
  index = buffer.indexOf('\n');
28
- if (Buffer.byteLength(line, 'utf8') > maxFrameBytes) {
37
+ if (Buffer.byteLength(line, 'utf8') + 1 > maxOutputFrameBytes) {
29
38
  fail(new Error('frame_too_large'));
30
39
  return;
31
40
  }
@@ -57,7 +66,7 @@ function createScriptSessionChannel(child, { maxFrameBytes = MAX_FRAME_BYTES } =
57
66
  }
58
67
  pending.push(message);
59
68
  }
60
- if (Buffer.byteLength(buffer, 'utf8') > maxFrameBytes) {
69
+ if (Buffer.byteLength(buffer, 'utf8') > maxOutputFrameBytes) {
61
70
  fail(new Error('frame_too_large'));
62
71
  }
63
72
  });
@@ -67,7 +76,7 @@ function createScriptSessionChannel(child, { maxFrameBytes = MAX_FRAME_BYTES } =
67
76
  function send(message) {
68
77
  if (closed || !child.stdin.writable) return;
69
78
  const line = `${JSON.stringify(message)}\n`;
70
- if (Buffer.byteLength(line, 'utf8') > maxFrameBytes) {
79
+ if (Buffer.byteLength(line, 'utf8') > maxInputFrameBytes) {
71
80
  fail(new Error('frame_too_large'));
72
81
  return;
73
82
  }
@@ -76,14 +85,11 @@ function createScriptSessionChannel(child, { maxFrameBytes = MAX_FRAME_BYTES } =
76
85
 
77
86
  function nextMessage() {
78
87
  return new Promise((resolve) => {
79
- if (closed) {
80
- resolve(failure);
81
- return;
82
- }
83
88
  if (pending.length > 0) {
84
89
  resolve(pending.shift());
85
90
  return;
86
91
  }
92
+ if (closed) { resolve(failure); return; }
87
93
  waiters.push(resolve);
88
94
  });
89
95
  }
@@ -108,7 +114,9 @@ function createScriptSessionChannel(child, { maxFrameBytes = MAX_FRAME_BYTES } =
108
114
  closed = true;
109
115
  const frame = { type: 'fail', error: error.message || String(error) };
110
116
  failure = frame;
117
+ const terminal = error.message === 'channel_closed' ? pending.find(message => ['return', 'fail'].includes(message.type)) : null;
111
118
  pending.length = 0;
119
+ if (terminal) pending.push(terminal);
112
120
  buffer = '';
113
121
  while (waiters.length > 0) waiters.shift()(frame);
114
122
  for (const resolve of replies.values()) resolve(frame);
@@ -121,7 +129,16 @@ function createScriptSessionChannel(child, { maxFrameBytes = MAX_FRAME_BYTES } =
121
129
  if (child.kill('SIGTERM') && !killTimer) killTimer = setTimeout(() => child.kill('SIGKILL'), 250);
122
130
  }
123
131
 
124
- return { send, nextMessage, waitReply, stop };
132
+ function diagnostics() {
133
+ if (!stderrTotalBytes) return {};
134
+ // Invalid bytes expand to replacement characters. Bound delivered UTF-8
135
+ // text too, starting at a complete character in the retained tail.
136
+ const text = Buffer.from(stderr.toString('utf8'));
137
+ let start = Math.max(0, text.length - maxDiagnosticBytes);
138
+ while (start < text.length && (text[start] & 0xc0) === 0x80) start++;
139
+ return { stderr: text.subarray(start).toString('utf8'), stderrTruncated: stderrTotalBytes > stderr.length || start > 0 };
140
+ }
141
+ return { send, nextMessage, waitReply, stop, diagnostics };
125
142
  }
126
143
 
127
144
  module.exports = { createScriptSessionChannel };
@@ -174,6 +174,7 @@ function createScriptSupervisor({
174
174
  mutationUnknown: false,
175
175
  decisionRevision: 0,
176
176
  currentRequestId: null,
177
+ pendingQuestion: null,
177
178
  questionRevisions: new Map(),
178
179
  decisions: new Map(),
179
180
  ledger: scriptLedger,
@@ -431,8 +432,10 @@ function createScriptSupervisor({
431
432
  operationId: record.operationId,
432
433
  status: record.status,
433
434
  pauseReason: record.pauseReason,
435
+ pendingQuestion: record.status === 'waiting_for_agent' ? record.pendingQuestion : null,
434
436
  error: record.error,
435
437
  hash: record.hash,
438
+ ...(record.diagnostics ? { diagnostics: record.diagnostics } : {}),
436
439
  resultRef: record.resultRef ?? null,
437
440
  eventSequence: record.events.sequence,
438
441
  events: record.events.after(args.afterSequence, args.eventLimit ?? args.limit),
@@ -517,6 +520,7 @@ function persistRecord(record, kind, body) {
517
520
  }
518
521
 
519
522
  async function commitTerminal(record, status, error, now, result) {
523
+ record.diagnostics = status === 'failed' ? result?.diagnostics : undefined;
520
524
  let revision = record.checkpointRevision + 1;
521
525
  let resultRef = null;
522
526
  if (status === 'completed' && record.store) {
@@ -539,6 +543,7 @@ async function commitTerminal(record, status, error, now, result) {
539
543
  status,
540
544
  error,
541
545
  resultRef,
546
+ ...(record.diagnostics ? { diagnostics: record.diagnostics } : {}),
542
547
  ...(record.recording ? { recording: record.recording.info() } : {}),
543
548
  };
544
549
  let persisted = await persistRecord(record, 'checkpoint', checkpoint);
@@ -777,6 +782,9 @@ function trackAgent(agent, record, now) {
777
782
  const requestId = `ask-${record.decisionRevision}`;
778
783
  record.questionRevisions.set(requestId, record.decisionRevision);
779
784
  record.currentRequestId = requestId;
785
+ // The unanswered question stays on the record so a filtered event page
786
+ // cannot hide it from status/wait.
787
+ record.pendingQuestion = { requestId, revision: record.decisionRevision, request };
780
788
  record.status = 'waiting_for_agent';
781
789
  emitRecord(record, 'agent_question_created', {
782
790
  request,
@@ -14,19 +14,16 @@ function validateValue(value, schema, field = '') {
14
14
  if (Object.hasOwn(schema, 'const') && value !== schema.const) throw invalid(field, `${label} must be ${JSON.stringify(schema.const)}.`);
15
15
  if (schema.enum && !schema.enum.includes(value)) throw invalid(field, `${label} must be one of: ${schema.enum.join(', ')}.`);
16
16
  const type = schema.type;
17
- const matches = type === undefined || (type === 'array' ? Array.isArray(value)
18
- : type === 'null' ? value === null
19
- : type === 'object' ? value !== null && typeof value === 'object' && !Array.isArray(value)
20
- : type === 'integer' ? Number.isSafeInteger(value)
21
- : type === 'number' ? typeof value === 'number' && Number.isFinite(value)
22
- : typeof value === type);
23
- if (!matches) throw invalid(field, `${label} must be ${type}; no implicit type conversion is performed.`);
17
+ if (!matchesType(value, type)) throw invalid(field, `${label} must be ${type}; no implicit type conversion is performed.`);
24
18
  if (schema.minLength !== undefined && value.length < schema.minLength) throw invalid(field, `${label} must not be empty.`);
25
19
  if (schema.maxLength !== undefined && value.length > schema.maxLength) throw invalid(field, `${label} must contain at most ${schema.maxLength} characters.`);
26
20
  if (schema.pattern !== undefined && !new RegExp(schema.pattern).test(value)) throw invalid(field, `${label} must match ${schema.pattern}.`);
27
- if (schema.minimum !== undefined && value < schema.minimum) throw invalid(field, `${label} must be >= ${schema.minimum}.`);
28
- if (schema.exclusiveMinimum !== undefined && value <= schema.exclusiveMinimum) throw invalid(field, `${label} must be > ${schema.exclusiveMinimum}.`);
29
- if (schema.maximum !== undefined && value > schema.maximum) throw invalid(field, `${label} must be <= ${schema.maximum}.`);
21
+ // A range violation quotes the field's documented meaning, so the caller can
22
+ // correct the call without reading the nested schema.
23
+ const explained = message => schema.description ? `${message} ${schema.description}` : message;
24
+ if (schema.minimum !== undefined && value < schema.minimum) throw invalid(field, explained(`${label} must be >= ${schema.minimum}.`));
25
+ if (schema.exclusiveMinimum !== undefined && value <= schema.exclusiveMinimum) throw invalid(field, explained(`${label} must be > ${schema.exclusiveMinimum}.`));
26
+ if (schema.maximum !== undefined && value > schema.maximum) throw invalid(field, explained(`${label} must be <= ${schema.maximum}.`));
30
27
  if (type === 'array') {
31
28
  if (schema.minItems !== undefined && value.length < schema.minItems) throw invalid(field, `${label} requires at least ${schema.minItems} item(s).`);
32
29
  if (schema.maxItems !== undefined && value.length > schema.maxItems) throw invalid(field, `${label} accepts at most ${schema.maxItems} item(s).`);
@@ -68,6 +65,15 @@ function validateValue(value, schema, field = '') {
68
65
  if (Object.keys(schema).every(key => ['description', 'default'].includes(key))) validateJson(value, field);
69
66
  }
70
67
 
68
+ function matchesType(value, type) {
69
+ return type === undefined || (type === 'array' ? Array.isArray(value)
70
+ : type === 'null' ? value === null
71
+ : type === 'object' ? value !== null && typeof value === 'object' && !Array.isArray(value)
72
+ : type === 'integer' ? Number.isSafeInteger(value)
73
+ : type === 'number' ? typeof value === 'number' && Number.isFinite(value)
74
+ : typeof value === type);
75
+ }
76
+
71
77
  function accepts(value, schema, field) {
72
78
  try { validateValue(value, schema, field); return true; }
73
79
  catch (error) { if (!(error instanceof CommandError)) throw error; return false; }
@@ -82,22 +88,48 @@ function validateUnion(value, branches, field, exactlyOne) {
82
88
  }
83
89
  if (matched && (!exactlyOne || matched === 1)) return;
84
90
  if (!matched) {
91
+ // A branch for another JSON type cannot explain this value, e.g. the null
92
+ // branch of a nullable target never explains an object missing platform.
93
+ const explained = failures.filter(({ branch }) => matchesType(value, branch.type));
94
+ if (!explained.length) {
95
+ throw invalid(field, `${field || 'arguments'} must be ${[...new Set(failures.map(({ branch }) => branch.type))].join(' or ')}; no implicit type conversion is performed.`);
96
+ }
85
97
  // A tagged branch gives a useful precise error instead of a generic union
86
98
  // failure, e.g. decision.action.durationMs for an observed long press.
87
- const tagged = failures.filter(({ branch }) => {
99
+ const tagged = explained.filter(({ branch }) => {
88
100
  const tags = Object.entries(branch.properties || {}).filter(([, rule]) => Object.hasOwn(rule, 'const'));
89
101
  return tags.some(([key, rule]) => value?.[key] === rule.const)
90
102
  && tags.every(([key, rule]) => Object.hasOwn(value || {}, key) ? value[key] === rule.const : !branch.required?.includes(key));
91
103
  });
92
104
  if (tagged.length === 1) throw tagged[0].error;
93
- const candidates = tagged.length ? tagged : failures;
105
+ const candidates = tagged.length ? tagged : explained;
94
106
  if (candidates.every(({ error }) => error.code === candidates[0].error.code && error.field === candidates[0].error.field && error.message === candidates[0].error.message)) throw candidates[0].error;
107
+ // No branch accepts every supplied tag: name the tag the intended variants
108
+ // reject and its accepted values instead of listing every variant.
109
+ const rejected = tagged.length ? null : rejectedDiscriminator(value || {}, explained.map(({ branch }) => branch));
110
+ if (rejected) {
111
+ const child = fieldAt(field, rejected.key);
112
+ throw invalid(child, `${child} must be one of: ${rejected.values.map(item => JSON.stringify(item)).join(', ')}.`);
113
+ }
95
114
  if (branches.length === 1) throw failures[0].error;
96
115
  }
97
116
  throw new CommandError('invalid_argument', `${field || 'arguments'} must match ${exactlyOne ? 'exactly one' : 'one'} of the documented variants.`,
98
117
  { field: field || 'arguments', details: { variants: failures.map(({ error }) => ({ field: error.field, error: error.code, message: error.message })) } });
99
118
  }
100
119
 
120
+ // The branches agreeing with the most supplied tags are the intended variants,
121
+ // e.g. the two start variants for {operation:"start", mode:"wrong"}. A supplied
122
+ // tag they all reject is the field to correct; its accepted values are theirs.
123
+ function rejectedDiscriminator(value, branches) {
124
+ const suppliedTags = branch => Object.entries(branch.properties || {}).filter(([key, rule]) => Object.hasOwn(rule, 'const') && Object.hasOwn(value, key));
125
+ const agreement = branch => suppliedTags(branch).filter(([key, rule]) => value[key] === rule.const).length;
126
+ const best = Math.max(...branches.map(agreement));
127
+ const closest = branches.filter(branch => agreement(branch) === best);
128
+ const key = suppliedTags(closest[0]).map(([name]) => name)
129
+ .find(name => closest.every(branch => Object.hasOwn(branch.properties[name] || {}, 'const') && value[name] !== branch.properties[name].const));
130
+ return key === undefined ? null : { key, values: [...new Set(closest.map(branch => branch.properties[key].const))] };
131
+ }
132
+
101
133
  function validateJson(value, field = '', ancestors = new Set(), depth = 0) {
102
134
  if (depth > 64) throw invalid(field, 'JSON nesting must not exceed 64 levels.');
103
135
  if (value === null || typeof value === 'string' || typeof value === 'boolean' || typeof value === 'number' && Number.isFinite(value)) return;
@@ -3,7 +3,7 @@
3
3
  const fs = require('node:fs');
4
4
  const path = require('node:path');
5
5
  const crypto = require('node:crypto');
6
- const { checksumOf, canonicalJson, validateArchivedRecord, verifyChecksum } = require('./evidence-schema');
6
+ const { checksumOf, canonicalJson, validateArchivedRecord, verifyChecksum, NAMESPACES } = require('./evidence-schema');
7
7
  const { analyzeRecordedPayloads } = require('./recorded-payload-archive');
8
8
  const { readRegularFile: readRecordedFile } = require('./evidence-recording');
9
9
 
@@ -23,7 +23,7 @@ function textArgument(value, field) {
23
23
  }
24
24
 
25
25
  function namespaceArgument(namespace) {
26
- requireValue(namespace === 'intent' || namespace === 'script', 'invalid_argument', 'namespace');
26
+ requireValue(NAMESPACES.includes(namespace), 'invalid_argument', 'namespace');
27
27
  }
28
28
 
29
29
  async function handle(args = {}, { getFactStore } = {}) {
@@ -3,7 +3,7 @@
3
3
  const crypto = require('node:crypto');
4
4
  const { sanitizePersistentValue } = require('../fact-codec');
5
5
 
6
- const NAMESPACES = Object.freeze(['script', 'intent']);
6
+ const NAMESPACES = Object.freeze(['script', 'intent', 'response']);
7
7
  const PROVIDERS = Object.freeze(['native', 'uia', 'flutter', 'h5']);
8
8
  const RECORD_SCHEMA = 'aab.execution-evidence/v1';
9
9
  const AGENT_DECISIONS = Object.freeze(['act', 'complete', 'fail', 'inconclusive']);
@@ -17,6 +17,7 @@ const KINDS = Object.freeze([
17
17
  'checkpoint',
18
18
  'result',
19
19
  'attachment',
20
+ 'response',
20
21
  ]);
21
22
 
22
23
  function checksumOf(value) {
@@ -57,6 +58,7 @@ function verifyChecksum(record) {
57
58
  }
58
59
 
59
60
  function requiredFields(kind) {
61
+ if (kind === 'response') return ['operationId', 'revision', 'snapshotBase64'];
60
62
  if (kind === 'observation') {
61
63
  return ['operationId', 'revision', 'target', 'provider', 'capturedAtMs'];
62
64
  }
@@ -97,6 +99,7 @@ function validateRecord(namespace, kind, record, { archivedUnversioned = false }
97
99
  if (!record || typeof record !== 'object') {
98
100
  return { ok: false, error: 'invalid_record' };
99
101
  }
102
+ if ((namespace === 'response') !== (kind === 'response')) return { ok: false, error: 'invalid_kind' };
100
103
  if (!archivedUnversioned && record.schemaVersion !== undefined && record.schemaVersion !== RECORD_SCHEMA) {
101
104
  return { ok: false, error: 'unsupported_record_schema', field: 'schemaVersion' };
102
105
  }
@@ -107,6 +110,18 @@ function validateRecord(namespace, kind, record, { archivedUnversioned = false }
107
110
  return { ok: false, error: 'invalid_record', field };
108
111
  }
109
112
  }
113
+ if (kind === 'response') {
114
+ const encoded = record.snapshotBase64;
115
+ if (typeof encoded !== 'string' || Buffer.from(encoded, 'base64').toString('base64') !== encoded
116
+ || record.revision !== 1) return { ok: false, error: 'invalid_record', field: 'snapshotBase64' };
117
+ try {
118
+ const snapshot = JSON.parse(Buffer.from(encoded, 'base64').toString('utf8'));
119
+ if (!['json', 'text'].includes(snapshot.kind) || !Object.hasOwn(snapshot, 'value')
120
+ || !snapshot.execution || !snapshot.control || snapshot.control.source !== undefined
121
+ || snapshot.identity?.responseId !== record.operationId || typeof snapshot.identity.command !== 'string'
122
+ || !Number.isSafeInteger(snapshot.identity.capturedAtMs)) throw new Error('invalid_snapshot');
123
+ } catch { return { ok: false, error: 'invalid_record', field: 'snapshotBase64' }; }
124
+ }
110
125
  if (kind === 'result' && (namespace !== 'script' || !Object.hasOwn(record, 'result')
111
126
  || !Number.isSafeInteger(record.bytes) || record.bytes < 1
112
127
  || !/^[a-f0-9]{64}$/.test(record.sha256) || !/^[a-f0-9]{64}$/.test(record.originalSha256)
@@ -1,10 +1,10 @@
1
1
  'use strict';
2
2
 
3
- const { buildEnvelope, verifyChecksum } = require('./evidence-schema');
3
+ const { buildEnvelope, verifyChecksum, NAMESPACES } = require('./evidence-schema');
4
4
 
5
5
  function createEvidenceStore({ namespace, adapter, now = Date.now, maxBytes = 8 * 1024 * 1024 } = {}) {
6
- if (namespace !== 'script' && namespace !== 'intent') {
7
- throw new TypeError('namespace must be script or intent');
6
+ if (!NAMESPACES.includes(namespace)) {
7
+ throw new TypeError('namespace must be script, intent or response');
8
8
  }
9
9
  if (!adapter || typeof adapter.record !== 'function' || typeof adapter.readById !== 'function') {
10
10
  throw new TypeError('adapter with record/readById is required');
@@ -133,8 +133,15 @@ function scriptSpecSchema(permissionNames) {
133
133
  }
134
134
 
135
135
  function executionCommandSchema(command, permissionNames) {
136
- const read = { afterSequence, limit: pageLimit };
137
- const scriptRead = { ...read, eventLimit: pageLimit, includeCatalog: boolean };
136
+ // Script and Intent page with different cursors; the schema states which one.
137
+ const historyRead = {
138
+ afterSequence: { ...afterSequence, description: 'History page cursor: the previous response\'s history.lastSequence. Not an eventSequence.' },
139
+ limit: { ...pageLimit, description: 'History entries per page; it bounds entries, not response bytes.' },
140
+ };
141
+ const scriptRead = {
142
+ afterSequence: { ...afterSequence, description: 'Continuation cursor: the previous status/wait response\'s eventSequence; events and history after it are returned. Not history.lastSequence.' },
143
+ limit: pageLimit, eventLimit: pageLimit, includeCatalog: boolean,
144
+ };
138
145
  const operation = (name, properties, required = []) => object({ operation: { const: name }, ...properties }, ['operation', ...required]);
139
146
  const control = (name, properties = {}) => operation(name, { operationId: text, ...properties }, ['operationId']);
140
147
  if (command === 'script') return variants('operation', [
@@ -144,7 +151,8 @@ function executionCommandSchema(command, permissionNames) {
144
151
  { if: { required: ['recordingDir'] }, then: { properties: { script: { properties: { policy: { properties: { restartPolicy: { const: 'none' } } } } } } } },
145
152
  ] },
146
153
  control('result'),
147
- control('status', scriptRead), control('wait', { ...scriptRead, waitMs: { ...integer(0, 60000), default: 30000 } }),
154
+ control('status', scriptRead), control('wait', { ...scriptRead, waitMs: { ...integer(0, 60000), default: 30000,
155
+ description: 'Upper bound of one wait call in milliseconds. To keep waiting, call wait again with the same operationId and afterSequence set to the last eventSequence.' } }),
148
156
  ...['pause', 'resume', 'cancel'].map(name => control(name, scriptRead)),
149
157
  operation('decide', { operationId: text, requestId: text, revision, decision: { contentMediaType: 'application/json', description: 'JSON answer to the current ctx.askAgent request; business data is preserved.' }, ...scriptRead }, ['operationId', 'requestId', 'revision', 'decision']),
150
158
  operation('runtime-status', {}),
@@ -156,7 +164,7 @@ function executionCommandSchema(command, permissionNames) {
156
164
  return variants('operation', [
157
165
  operation('start', { ...start, mode: { const: 'supervised', default: 'supervised' } }, ['goal', 'target']),
158
166
  operation('start', { ...start, mode: { const: 'autonomous' }, agentModule: text, budget: intentBudget }, ['goal', 'target', 'mode', 'agentModule']),
159
- control('status', read), control('observe', { basedOnRevision: revision, observationTarget: observationTargetSchema,
167
+ control('status', historyRead), control('observe', { basedOnRevision: revision, observationTarget: observationTargetSchema,
160
168
  provider: { ...provider, description: 'Observe through this provider on the frozen target. Becomes selected only after the new observation and summary are committed; omitted retains the last committed selection.' } }),
161
169
  operation('decide', { operationId: text, decision: intentDecisionSchema() }, ['operationId', 'decision']),
162
170
  ...['pause', 'resume', 'cancel'].map(name => control(name)),
@@ -164,7 +172,7 @@ function executionCommandSchema(command, permissionNames) {
164
172
  ]);
165
173
  }
166
174
  if (command === 'evidence') return variants('operation', [
167
- operation('export', { namespace: { enum: ['intent', 'script'] }, operationId: text, outputDir: text, includeRecordedPayloads: boolean }, ['namespace', 'operationId', 'outputDir']),
175
+ operation('export', { namespace: { enum: ['intent', 'script', 'response'] }, operationId: text, outputDir: text, includeRecordedPayloads: boolean }, ['namespace', 'operationId', 'outputDir']),
168
176
  operation('verify', { archiveDir: text, manifestSha256: { type: 'string', pattern: '^[a-f0-9]{64}$' } }, ['archiveDir', 'manifestSha256']),
169
177
  ]);
170
178
  throw new Error(`No execution contract for ${command}`);
@@ -73,6 +73,10 @@ function commandPlatform(command) {
73
73
  // select another platform; a partial identity can override defaults only on the
74
74
  // same platform. Never mix an Android serial with an iOS/Web connection.
75
75
  function bindCommandTarget(command, args, defaultTarget = null, explicitTarget) {
76
+ if (command === 'executor-prepare') {
77
+ if (explicitTarget !== undefined) throw new CommandError('target_platform_mismatch', 'Executor preparation runs on the Host and accepts no device target.');
78
+ return { target: null, args: { ...args } };
79
+ }
76
80
  const platform = commandPlatform(command);
77
81
  const definition = definitions[platform];
78
82
  const selected = explicitTarget === undefined ? defaultTarget : normalizeExecutionTarget(explicitTarget);
@@ -7,8 +7,8 @@ const context = new AsyncLocalStorage();
7
7
  const requestDirectory = () => context.getStore()?.cwd ?? process.cwd();
8
8
  const resolveRequestPath = value => path.resolve(requestDirectory(), value);
9
9
  const inRequestDirectory = (cwd, action) => context.run({ cwd }, action);
10
- const localPaths = ['outFile', 'artifactDir', 'apkPath', 'appPath', 'recordingDir', 'outputDir', 'archiveDir', 'wdaProjectPath', 'agentModule'];
11
- const executables = ['adb', 'aaptPath', 'apksignerPath', 'devicectl', 'xcodebuild', 'pythonPath'];
10
+ const localPaths = ['outFile', 'artifactDir', 'apkPath', 'appPath', 'recordingDir', 'outputDir', 'archiveDir', 'wdaProjectPath', 'agentModule', 'projectDir', 'testPackagePath'];
11
+ const executables = ['adb', 'aaptPath', 'apksignerPath', 'devicectl', 'xcodebuild', 'pythonPath', 'flutterPath', 'gradlePath'];
12
12
 
13
13
  // Only contract-defined filesystem fields are resolved. Web URL paths, source
14
14
  // text and arbitrary Script inputs keep their original values.
@@ -165,6 +165,8 @@ function requestDigest(command, args) {
165
165
  }
166
166
 
167
167
  function targetFor(command, args) {
168
+ if (command === 'executor-prepare') return { kind: 'host', platform: 'host',
169
+ key: `executor-prepare:${JSON.stringify([args.platform, args.projectDir ?? null])}` };
168
170
  if (command === 'logcat' && String(args.deviceLogScope || '').toLowerCase() === 'device') {
169
171
  const serial = targetPart(args.serial);
170
172
  return {
@@ -13,7 +13,7 @@ a light command directory. Use `capabilities {"command":"tap-text"}`
13
13
  for its current `inputSchema`, platform, role and supported entrypoints. Domain
14
14
  `execution` contains Intent, Script, runtime lifecycle and device ownership;
15
15
  `evidence` contains archive operations.
16
- `capabilities {"includeOptions":true}` returns every current command schema.
16
+ Discovery has a fixed 96 KiB budget. Broad `includeOptions:true` queries may return `discovery_output_too_large`; use the returned narrower query. Schemas are never silently pruned.
17
17
 
18
18
  Load only the operation needed for Intent, Script or evidence, for example
19
19
  `capabilities {"command":"intent","operation":"start"}`. To inspect an Intent
@@ -24,10 +24,19 @@ runtime contract. Intent terminal decisions remain available in the selected
24
24
  decision schema. Unsupported operations or scope combinations return an error
25
25
  with the offending field. Omit filters to read the complete command contract.
26
26
 
27
- All parameters are under `run.arguments`:
27
+ Business parameters are under `run.arguments`. The required top-level `extract` chooses delivery; explicit `null` requests the original result within budget:
28
28
 
29
29
  ```json
30
- {"command":"tap-text","arguments":{"serial":"DEVICE","packageName":"com.example.app","targetText":"设置","provider":"auto"}}
30
+ {
31
+ "command": "tap-text",
32
+ "extract": null,
33
+ "arguments": {
34
+ "serial": "DEVICE",
35
+ "packageName": "com.example.app",
36
+ "targetText": "设置",
37
+ "provider": "auto"
38
+ }
39
+ }
31
40
  ```
32
41
 
33
42
  Names are canonical and case-sensitive. Unknown arguments, aliases, invalid
@@ -55,8 +64,8 @@ provider access and device ownership through their execution-specific contracts.
55
64
  ### Shared runtime lifecycle
56
65
 
57
66
  The first executing command starts a local runtime. `runtime --operation start`
58
- starts it explicitly; `runtime --operation status` inspects it without starting
59
- one; `runtime --operation stop` cancels and drains active work before releasing
67
+ starts it explicitly; `runtime --operation status --extract null` inspects it without starting
68
+ one; `runtime --operation stop --extract null` cancels and drains active work before releasing
60
69
  its ownership. CLI exit, MCP EOF and client SIGINT/SIGTERM close only that client.
61
70
  Use `intent`/`script --operation cancel --operation-id ID` to cancel one task.
62
71
  The same operation ID can be queried and controlled from either entrypoint.
@@ -93,12 +102,26 @@ there; Script freezes its `cwd`, source and target before starting. Subsequent
93
102
  clients cannot change those paths. `evidence verify` is offline in both adapters
94
103
  and requires neither a running runtime nor an available FactStore.
95
104
 
96
- CLI results are one JSON envelope: `{kind: "json"|"text"|"bytes", value, history?}`.
97
- JSON false, zero and null remain values; text is a string and bytes are base64.
98
- `history` contains the same recording outcome that MCP exposes as `_history`.
99
- Command failures have `value.ok:false` and exit code 1. MCP uses its normal
100
- content and `isError` representation. These are wire-format differences;
101
- validation, dispatch, task control and evidence semantics are shared.
105
+ CLI results are one compact JSON envelope:
106
+ `{command, execution, control, extraction, delivery, kind, value?, failureStage?}`.
107
+ MCP returns this same body in tool text; no second `_meta`, `_history` or
108
+ `structuredContent` copy is attached. `execution` records the command outcome;
109
+ `control` retains continuation, current pending questions, receipts and capture
110
+ coverage. With `extract:null`, `value` is the command's original value including
111
+ `_feedback`; otherwise it is the extracted JSON value when delivered.
112
+
113
+ The default final-body budget is 96 KiB UTF-8. `output.maxBytes` can select
114
+ 16–256 KiB. Overflow never returns a truncated JSON document or a large original
115
+ fallback. Inspect `control.source`: only `persisted:true` provides a readable
116
+ ref. Retry extraction with `response read` and this ref; repeating the original
117
+ action with the same requestId is not a recovery guarantee. Original response
118
+ snapshots and their exports preserve the original JSON representation.
119
+
120
+ Exit 0 means requested delivery succeeded; 1 means validation/execution failed
121
+ or remained unknown; 2 means the command succeeded but extraction/delivery
122
+ failed. `failureStage` prioritizes validation, execution, extraction, delivery.
123
+ MCP sets `isError` consistently. See [response extraction](RESPONSE_EXTRACTION.md)
124
+ for exact examples, budgets, ref recovery and the distinction from Script.
102
125
 
103
126
  `batch`, `smoke`, `launch-native-test` and `launch-flutter` were removed. Use a
104
127
  code Script for sequences, `launch-app` for a Flutter app's actual launcher, and
@@ -503,6 +526,30 @@ or call budget prevents another Agent call. Malformed replies remain
503
526
  `waiting_for_decision` with a field error and require explicit correction; they
504
527
  neither dispatch nor trigger another Agent request automatically.
505
528
 
529
+ A supervised Intent is driven by the caller: observe, decide against the
530
+ observed revision, observe again, judge independently. One complete lifecycle,
531
+ as sent to `run` (the repository test suite runs this sequence against the
532
+ current contract with an injected device):
533
+
534
+ ```json lifecycle-example
535
+ {"command":"intent","extract":null,"arguments":{"operation":"start","goal":"Open the Labels screen","target":{"platform":"android","serial":"<serial>","packageName":"<package>"}}}
536
+ {"command":"intent","extract":null,"arguments":{"operation":"decide","operationId":"<operationId>","decision":{"decisionId":"d-1","agentDecision":"act","basedOnRevision":1,"action":{"action":"tap","selector":{"text":"Labels"}}}}}
537
+ {"command":"intent","extract":null,"arguments":{"operation":"status","operationId":"<operationId>","limit":20}}
538
+ {"command":"intent","extract":null,"arguments":{"operation":"status","operationId":"<operationId>","limit":20,"afterSequence":"<history.lastSequence of the previous page>"}}
539
+ {"command":"intent","extract":null,"arguments":{"operation":"decide","operationId":"<operationId>","decision":{"decisionId":"d-2","agentDecision":"complete","basedOnRevision":2}}}
540
+ ```
541
+
542
+ `start` and each `decide` return the committed observation with its `revision`
543
+ and `summary`; the next decision must carry that `revision` as
544
+ `basedOnRevision`. `status` returns the current state plus one history page.
545
+ `limit` bounds entries, not bytes: a page of `summary` entries can still be
546
+ large, so read the current state with a small `limit` and page history
547
+ deliberately. The next page's `afterSequence` is the previous page's
548
+ `history.lastSequence`; it is not an `eventSequence`, which belongs to Script.
549
+ Supervised mode never acts on its own: without a decision the Intent stays
550
+ `waiting_for_decision` until `timeoutMs`, and `complete` records the caller's
551
+ judgment rather than evidence of the business outcome.
552
+
506
553
  Every decision requires `decisionId`, positive integer `basedOnRevision`, and
507
554
  `agentDecision`. `act` requires a provider-specific `action`; terminal decisions
508
555
  `complete/fail/inconclusive` forbid one. Tap uses an exact `selector` object, with
@@ -1186,7 +1233,23 @@ exact and come from one fresh visible provider tree; metadata, hidden windows
1186
1233
  and text from another provider cannot complete a condition. For example:
1187
1234
 
1188
1235
  ```json
1189
- {"command":"wait-text","arguments":{"serial":"DEVICE","packageName":"com.example.app","provider":"native","targetText":"Save, draft","requireText":["Editor"],"absentText":["Loading"],"timeoutMs":5000}}
1236
+ {
1237
+ "command": "wait-text",
1238
+ "extract": null,
1239
+ "arguments": {
1240
+ "serial": "DEVICE",
1241
+ "packageName": "com.example.app",
1242
+ "provider": "native",
1243
+ "targetText": "Save, draft",
1244
+ "requireText": [
1245
+ "Editor"
1246
+ ],
1247
+ "absentText": [
1248
+ "Loading"
1249
+ ],
1250
+ "timeoutMs": 5000
1251
+ }
1252
+ }
1190
1253
  ```
1191
1254
 
1192
1255
  At least one condition is required. Pure absence or Activity-only waits require
@@ -1491,7 +1554,7 @@ The operation holds the phone's mutation lease through package verification.
1491
1554
  For example, after receiving a fresh observation:
1492
1555
 
1493
1556
  ```json
1494
- {"command":"intent","arguments":{"operation":"decide","operationId":"FROM_START","decision":{"decisionId":"choice-1","basedOnRevision":2,"agentDecision":"act","action":{"action":"tap","selector":{"resourceName":"ID_FROM_ACTUAL_OBSERVATION"}}}}}
1557
+ {"command":"intent","extract":null,"arguments":{"operation":"decide","operationId":"FROM_START","decision":{"decisionId":"choice-1","basedOnRevision":2,"agentDecision":"act","action":{"action":"tap","selector":{"resourceName":"ID_FROM_ACTUAL_OBSERVATION"}}}}}
1495
1558
  ```
1496
1559
 
1497
1560
  Completion requires the matching original shell-job receipt, a successful
@@ -1535,11 +1598,11 @@ provider operation. A history failure never retries or rewrites that operation.
1535
1598
  Intent/Script required evidence commits retain their existing strict admission
1536
1599
  and terminal-evidence contracts.
1537
1600
 
1538
- Object results carry `_history` with schema `aab.command-history/v1`, including
1539
- when `feedback:"off"`. Every ordinary CLI/MCP result also carries the same value in
1540
- `_meta["ai-app-bridge/history"]`, so raw text/image-result consumers can inspect
1541
- history without rewriting the original content. This is MCP history metadata;
1542
- CLI and MCP use the same history store; direct internal provider calls do not acquire an auxiliary one.
1601
+ Public replies carry command recording status in `control.history` with schema
1602
+ `aab.command-history/v1`, where applicable. Full Intent history and original
1603
+ business values remain in value; only page/continuation fields are protected.
1604
+ CLI and MCP share the same compact body. There is no duplicate `_history` or
1605
+ MCP `_meta["ai-app-bridge/history"]` attachment.
1543
1606
 
1544
1607
  - `stored`: this invocation's action and returned Host evidence references were
1545
1608
  committed. It is not a claim of complete mobile history or business correctness.
@@ -1562,7 +1625,16 @@ Background observer health remains separate from this foreground write report.
1562
1625
  Trigger the runtime permission request through the App's existing flow, then call:
1563
1626
 
1564
1627
  ```json
1565
- {"command":"permission-dialog","arguments":{"serial":"DEVICE","packageName":"com.example.app","permission":"android.permission.RECORD_AUDIO","outcome":"allow-once"}}
1628
+ {
1629
+ "command": "permission-dialog",
1630
+ "extract": null,
1631
+ "arguments": {
1632
+ "serial": "DEVICE",
1633
+ "packageName": "com.example.app",
1634
+ "permission": "android.permission.RECORD_AUDIO",
1635
+ "outcome": "allow-once"
1636
+ }
1637
+ }
1566
1638
  ```
1567
1639
 
1568
1640
  This MCP entry returns an ordinary Intent ID, the actual UI summary, and
@@ -1,7 +1,7 @@
1
1
  # Record, export and verify execution evidence
2
2
 
3
3
  Discover the public MCP command with `capabilities {"command":"evidence"}`.
4
- It works for both Intent operation IDs and Script execution IDs, including
4
+ It works for Intent operation IDs, Script execution IDs and saved response IDs, including
5
5
  records retained after the execution runtime has restarted. CLI and MCP use the
6
6
  same operation IDs and export command. Export does not require a
7
7
  connected phone or a live worker.
@@ -9,6 +9,7 @@ connected phone or a live worker.
9
9
  ```json
10
10
  {
11
11
  "command": "evidence",
12
+ "extract": null,
12
13
  "arguments": {
13
14
  "operation": "export",
14
15
  "namespace": "intent",
@@ -24,6 +25,16 @@ queued writes, and freezes the retained records for exactly that namespace
24
25
  and operation. The parent directory must exist; the output directory must
25
26
  not exist. No existing directory is replaced.
26
27
 
28
+ For a saved public response, use `namespace: "response"` and the unchanged
29
+ `control.source.ref.operationId`. A response record stores the final command
30
+ result and execution/control facts in `snapshotBase64`, using canonical UTF-8
31
+ JSON bytes after feedback has been added. The decoded bytes are exactly the
32
+ input used for extraction. They can differ from a UI history record captured
33
+ earlier. Export and verification preserve these bytes and use the existing
34
+ evidence checksum; no second snapshot checksum or archive format is added.
35
+ Only `control.source.persisted: true` supplies a readable ref. Missing, expired
36
+ or evicted responses fail reads; they never cause a device query or action.
37
+
27
38
  The response includes `archiveDir`, `manifestPath`, `manifestSha256`,
28
39
  `recordCount`, `targets`, and `coverage`. Save the returned manifest SHA256
29
40
  separately when handing the archive to an author or reviewer.
@@ -52,6 +63,7 @@ Move or copy the directory as a unit, then call:
52
63
  ```json
53
64
  {
54
65
  "command": "evidence",
66
+ "extract": null,
55
67
  "arguments": {
56
68
  "operation": "verify",
57
69
  "archiveDir": "/absolute/moved-archive",
@@ -181,6 +193,7 @@ Use the same public export call with `includeRecordedPayloads: true`:
181
193
  ```json
182
194
  {
183
195
  "command": "evidence",
196
+ "extract": null,
184
197
  "arguments": {
185
198
  "operation": "export",
186
199
  "namespace": "script",