@midscene/core 1.10.2-beta-20260706032158.0 → 1.10.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/es/agent/agent.mjs +40 -1
- package/dist/es/agent/agent.mjs.map +1 -1
- package/dist/es/agent/metrics.mjs +81 -0
- package/dist/es/agent/metrics.mjs.map +1 -0
- package/dist/es/agent/task-builder.mjs +1 -1
- package/dist/es/agent/task-builder.mjs.map +1 -1
- package/dist/es/agent/utils.mjs +1 -1
- package/dist/es/ai-model/connectivity/run-connectivity-test.mjs +9 -3
- package/dist/es/ai-model/connectivity/run-connectivity-test.mjs.map +1 -1
- package/dist/es/ai-model/inspect.mjs +3 -3
- package/dist/es/ai-model/inspect.mjs.map +1 -1
- package/dist/es/ai-model/llm-planning.mjs +1 -1
- package/dist/es/ai-model/llm-planning.mjs.map +1 -1
- package/dist/es/ai-model/model-adapter/resolve.mjs +2 -2
- package/dist/es/ai-model/model-adapter/resolve.mjs.map +1 -1
- package/dist/es/ai-model/models/default.mjs +2 -1
- package/dist/es/ai-model/models/default.mjs.map +1 -1
- package/dist/es/ai-model/models/doubao.mjs +38 -60
- package/dist/es/ai-model/models/doubao.mjs.map +1 -1
- package/dist/es/ai-model/models/gemini.mjs +4 -15
- package/dist/es/ai-model/models/gemini.mjs.map +1 -1
- package/dist/es/ai-model/models/glm.mjs +4 -0
- package/dist/es/ai-model/models/glm.mjs.map +1 -1
- package/dist/es/ai-model/models/gpt.mjs +5 -1
- package/dist/es/ai-model/models/gpt.mjs.map +1 -1
- package/dist/es/ai-model/models/kimi.mjs +4 -0
- package/dist/es/ai-model/models/kimi.mjs.map +1 -1
- package/dist/es/ai-model/models/mimo.mjs +3 -2
- package/dist/es/ai-model/models/mimo.mjs.map +1 -1
- package/dist/es/ai-model/models/qwen.mjs.map +1 -1
- package/dist/es/ai-model/models/ui-tars/adapter.mjs +2 -41
- package/dist/es/ai-model/models/ui-tars/adapter.mjs.map +1 -1
- package/dist/es/ai-model/models/utils/intent.mjs +6 -0
- package/dist/es/ai-model/models/utils/intent.mjs.map +1 -0
- package/dist/es/ai-model/prompt/describe.mjs +31 -34
- package/dist/es/ai-model/prompt/describe.mjs.map +1 -1
- package/dist/es/ai-model/prompt/extraction.mjs +15 -9
- package/dist/es/ai-model/prompt/extraction.mjs.map +1 -1
- package/dist/es/ai-model/prompt/llm-planning.mjs +19 -19
- package/dist/es/ai-model/prompt/llm-planning.mjs.map +1 -1
- package/dist/es/ai-model/prompt/locate-grounding-rules.mjs +7 -3
- package/dist/es/ai-model/prompt/locate-grounding-rules.mjs.map +1 -1
- package/dist/es/ai-model/prompt/recorder-metadata-generator.mjs +2 -1
- package/dist/es/ai-model/prompt/recorder-metadata-generator.mjs.map +1 -1
- package/dist/es/ai-model/service-caller/index.mjs +40 -19
- package/dist/es/ai-model/service-caller/index.mjs.map +1 -1
- package/dist/es/ai-model/service-caller/json.mjs +31 -28
- package/dist/es/ai-model/service-caller/json.mjs.map +1 -1
- package/dist/es/device/index.mjs +2 -4
- package/dist/es/device/index.mjs.map +1 -1
- package/dist/es/dump/html-utils.mjs +63 -1
- package/dist/es/dump/html-utils.mjs.map +1 -1
- package/dist/es/dump/screenshot-restoration.mjs +6 -0
- package/dist/es/dump/screenshot-restoration.mjs.map +1 -1
- package/dist/es/element-describer.mjs +36 -18
- package/dist/es/element-describer.mjs.map +1 -1
- package/dist/es/index.mjs +2 -2
- package/dist/es/index.mjs.map +1 -1
- package/dist/es/report-generator.mjs +28 -1
- package/dist/es/report-generator.mjs.map +1 -1
- package/dist/es/report-markdown.mjs +300 -35
- package/dist/es/report-markdown.mjs.map +1 -1
- package/dist/es/report.mjs +29 -1
- package/dist/es/report.mjs.map +1 -1
- package/dist/es/service/index.mjs +101 -25
- package/dist/es/service/index.mjs.map +1 -1
- package/dist/es/service/utils.mjs +65 -1
- package/dist/es/service/utils.mjs.map +1 -1
- package/dist/es/types.mjs.map +1 -1
- package/dist/es/utils.mjs +2 -2
- package/dist/lib/agent/agent.js +40 -1
- package/dist/lib/agent/agent.js.map +1 -1
- package/dist/lib/agent/metrics.js +115 -0
- package/dist/lib/agent/metrics.js.map +1 -0
- package/dist/lib/agent/task-builder.js +1 -1
- package/dist/lib/agent/task-builder.js.map +1 -1
- package/dist/lib/agent/utils.js +1 -1
- package/dist/lib/ai-model/connectivity/run-connectivity-test.js +9 -3
- package/dist/lib/ai-model/connectivity/run-connectivity-test.js.map +1 -1
- package/dist/lib/ai-model/inspect.js +3 -3
- package/dist/lib/ai-model/inspect.js.map +1 -1
- package/dist/lib/ai-model/llm-planning.js +1 -1
- package/dist/lib/ai-model/llm-planning.js.map +1 -1
- package/dist/lib/ai-model/model-adapter/resolve.js +1 -1
- package/dist/lib/ai-model/model-adapter/resolve.js.map +1 -1
- package/dist/lib/ai-model/models/default.js +2 -1
- package/dist/lib/ai-model/models/default.js.map +1 -1
- package/dist/lib/ai-model/models/doubao.js +38 -69
- package/dist/lib/ai-model/models/doubao.js.map +1 -1
- package/dist/lib/ai-model/models/gemini.js +4 -15
- package/dist/lib/ai-model/models/gemini.js.map +1 -1
- package/dist/lib/ai-model/models/glm.js +4 -0
- package/dist/lib/ai-model/models/glm.js.map +1 -1
- package/dist/lib/ai-model/models/gpt.js +5 -1
- package/dist/lib/ai-model/models/gpt.js.map +1 -1
- package/dist/lib/ai-model/models/kimi.js +4 -0
- package/dist/lib/ai-model/models/kimi.js.map +1 -1
- package/dist/lib/ai-model/models/mimo.js +3 -2
- package/dist/lib/ai-model/models/mimo.js.map +1 -1
- package/dist/lib/ai-model/models/qwen.js.map +1 -1
- package/dist/lib/ai-model/models/ui-tars/adapter.js +1 -40
- package/dist/lib/ai-model/models/ui-tars/adapter.js.map +1 -1
- package/dist/lib/ai-model/models/utils/intent.js +40 -0
- package/dist/lib/ai-model/models/utils/intent.js.map +1 -0
- package/dist/lib/ai-model/prompt/describe.js +31 -34
- package/dist/lib/ai-model/prompt/describe.js.map +1 -1
- package/dist/lib/ai-model/prompt/extraction.js +14 -8
- package/dist/lib/ai-model/prompt/extraction.js.map +1 -1
- package/dist/lib/ai-model/prompt/llm-planning.js +19 -19
- package/dist/lib/ai-model/prompt/llm-planning.js.map +1 -1
- package/dist/lib/ai-model/prompt/locate-grounding-rules.js +7 -3
- package/dist/lib/ai-model/prompt/locate-grounding-rules.js.map +1 -1
- package/dist/lib/ai-model/prompt/recorder-metadata-generator.js +4 -3
- package/dist/lib/ai-model/prompt/recorder-metadata-generator.js.map +1 -1
- package/dist/lib/ai-model/service-caller/index.js +44 -23
- package/dist/lib/ai-model/service-caller/index.js.map +1 -1
- package/dist/lib/ai-model/service-caller/json.js +33 -33
- package/dist/lib/ai-model/service-caller/json.js.map +1 -1
- package/dist/lib/device/index.js +2 -4
- package/dist/lib/device/index.js.map +1 -1
- package/dist/lib/dump/html-utils.js +65 -0
- package/dist/lib/dump/html-utils.js.map +1 -1
- package/dist/lib/dump/screenshot-restoration.js +6 -0
- package/dist/lib/dump/screenshot-restoration.js.map +1 -1
- package/dist/lib/element-describer.js +37 -22
- package/dist/lib/element-describer.js.map +1 -1
- package/dist/lib/index.js +1 -4
- package/dist/lib/index.js.map +1 -1
- package/dist/lib/report-generator.js +27 -0
- package/dist/lib/report-generator.js.map +1 -1
- package/dist/lib/report-markdown.js +300 -35
- package/dist/lib/report-markdown.js.map +1 -1
- package/dist/lib/report.js +28 -0
- package/dist/lib/report.js.map +1 -1
- package/dist/lib/service/index.js +99 -23
- package/dist/lib/service/index.js.map +1 -1
- package/dist/lib/service/utils.js +98 -1
- package/dist/lib/service/utils.js.map +1 -1
- package/dist/lib/types.js.map +1 -1
- package/dist/lib/utils.js +2 -2
- package/dist/types/agent/agent.d.ts +15 -2
- package/dist/types/agent/index.d.ts +1 -0
- package/dist/types/agent/metrics.d.ts +48 -0
- package/dist/types/ai-model/model-adapter/types.d.ts +8 -0
- package/dist/types/ai-model/models/doubao.d.ts +1 -4
- package/dist/types/ai-model/models/gemini.d.ts +1 -1
- package/dist/types/ai-model/models/registry.d.ts +1 -1
- package/dist/types/ai-model/models/utils/intent.d.ts +2 -0
- package/dist/types/ai-model/service-caller/index.d.ts +8 -3
- package/dist/types/ai-model/service-caller/json.d.ts +19 -2
- package/dist/types/device/device-options.d.ts +0 -28
- package/dist/types/device/index.d.ts +0 -5
- package/dist/types/dump/html-utils.d.ts +2 -0
- package/dist/types/dump/screenshot-restoration.d.ts +1 -1
- package/dist/types/element-describer.d.ts +3 -6
- package/dist/types/index.d.ts +2 -2
- package/dist/types/report-generator.d.ts +6 -0
- package/dist/types/service/index.d.ts +1 -1
- package/dist/types/service/utils.d.ts +21 -2
- package/dist/types/types.d.ts +12 -2
- package/package.json +3 -3
|
@@ -21,53 +21,50 @@ const getExamples = (language)=>{
|
|
|
21
21
|
const examples = examplesMap[language] || examplesMap.English;
|
|
22
22
|
return examples.map((e)=>`- ${e}`).join('\n');
|
|
23
23
|
};
|
|
24
|
+
const describeElementReturnJsonSchema = ()=>`{
|
|
25
|
+
"description": "unique element identifier",
|
|
26
|
+
"error"?: "error message if any"
|
|
27
|
+
}`;
|
|
24
28
|
const elementDescriberInstruction = ()=>{
|
|
25
29
|
const preferredLanguage = getPreferredLanguage();
|
|
26
30
|
return `
|
|
27
|
-
Describe the
|
|
31
|
+
Describe the real page element indicated by the temporary callout.
|
|
32
|
+
The callout is an annotation overlay. It is not part of the page or target.
|
|
33
|
+
The description will be used later to locate the same element on the original screenshot without annotations, so write a locator-style description instead of a visual caption.
|
|
28
34
|
|
|
29
35
|
IMPORTANT: You MUST write the description in ${preferredLanguage}.
|
|
30
36
|
|
|
31
|
-
|
|
32
|
-
1.
|
|
33
|
-
2.
|
|
34
|
-
3.
|
|
35
|
-
|
|
36
|
-
DESCRIPTION STRUCTURE:
|
|
37
|
-
1. Element type (button, input, link, div, etc.)
|
|
38
|
-
2. Primary identifier (in order of preference):
|
|
39
|
-
- Unique text content: "with text 'Login'"
|
|
40
|
-
- Unique attribute: "with aria-label 'Search'"
|
|
41
|
-
- Unique class/ID: "with class 'primary-button'"
|
|
42
|
-
- Unique position: "in header navigation"
|
|
43
|
-
3. Secondary identifiers (if needed for uniqueness):
|
|
44
|
-
- Visual features: "blue background", "with icon"
|
|
45
|
-
- Relative position: "below search bar", "in sidebar"
|
|
46
|
-
- Parent context: "in login form", "in main menu"
|
|
47
|
-
- Neighboring stable text: "to the right of the 'Settings' section title"
|
|
37
|
+
OBSERVE IN THIS ORDER:
|
|
38
|
+
1. Target first: identify the smallest real UI part at the callout endpoint/center: text, glyph, icon, arrow, input, dropdown/select, option, button, link, status, checkbox, radio, switch, tab, menu item, slider, image, control, or empty region.
|
|
39
|
+
2. Primitive: name what that smallest part is before adding surrounding context.
|
|
40
|
+
3. Owner/context: add the nearest stable owner only when it helps disambiguate, such as a label, row/card title, column header, field name, or adjacent visible text.
|
|
41
|
+
4. Similar candidates: if multiple candidates look similar, add stable local anchors from the same row, card, field, header, or group. Prefer visible text and values over inferred row counting or temporary visual state.
|
|
48
42
|
|
|
49
|
-
|
|
50
|
-
- Keep description under
|
|
51
|
-
-
|
|
52
|
-
-
|
|
53
|
-
-
|
|
54
|
-
-
|
|
55
|
-
-
|
|
56
|
-
- If the
|
|
57
|
-
-
|
|
58
|
-
-
|
|
59
|
-
-
|
|
60
|
-
-
|
|
43
|
+
RULES:
|
|
44
|
+
- Keep description under 35 words.
|
|
45
|
+
- Describe the smallest indicated UI part itself, not the larger container, row, card, sentence, or group that merely contains it.
|
|
46
|
+
- Ignore every annotation overlay, including callout number, line, color, marker, border, dot, ring, crosshair, or selection box. Never describe the annotation as the target.
|
|
47
|
+
- Do not borrow the text, glyph, direction, purpose, or state from a nearby element outside the callout endpoint/center.
|
|
48
|
+
- For tiny or icon-only controls, name the visible glyph/control and add its owner/context; adjacent text is context, not the target. If similar tiny controls are adjacent in the same group, add local order or relative position inside that group.
|
|
49
|
+
- If the endpoint/center is on a field value, label, or input body, describe that value/field/control. Do not retarget to a trailing icon, dropdown arrow, clear button, or search affordance unless the endpoint/center is on that icon itself.
|
|
50
|
+
- If the endpoint/center is inside the bordered body, current value, trigger, or blank area of a select/dropdown/combobox/filter field, use primitive "dropdown" and describe that dropdown/select control. Do not call it an input unless it is clearly a free-text field.
|
|
51
|
+
- If the endpoint/center is inside the bordered body or blank area of an input or filter field, describe the field body/current value/control even when the visible text is not exactly under the endpoint. Use the field label or visible value as context; do not snap to trailing icons or nearby table headers.
|
|
52
|
+
- If the endpoint/center is on an expanded dropdown/select/menu list item, use primitive "option" for selectable list options or "menuitem" for command menu entries.
|
|
53
|
+
- Only use primitive "icon" or "arrow" when the endpoint/center directly overlaps the real glyph strokes. A nearby search icon, dropdown arrow, clear button, or wrong locator result must not become the target primitive.
|
|
54
|
+
- For compound controls or stacked glyphs, describe only the sub-part containing the callout endpoint/center, using upper/lower or left/right only when visible.
|
|
55
|
+
- For inline text, links, or substrings, describe only the exact substring/link at the endpoint/center, not the whole line.
|
|
56
|
+
- For repeated rows, cards, or options, use same-local anchors that are visible in the screenshot, such as neighboring cell text, field value, title, date, time, ID, or column/header label.
|
|
57
|
+
- For tables or grids, describe the target as the intersection of the target column/header and same-row anchors. Do not use row ordinals or column ordinals unless the index/header is clearly visible.
|
|
58
|
+
- Use selected, highlighted, hovered, focused, or active state only if the callout endpoint/center is inside that state.
|
|
59
|
+
- If the endpoint/center is on blank space, describe the empty region/gap between stable surrounding anchors. Do not invent a nearby control.
|
|
60
|
+
- Use actual visible text from the current screenshot when available; do not copy labels from the examples.
|
|
61
61
|
- **Write the description in ${preferredLanguage}**
|
|
62
62
|
|
|
63
63
|
EXAMPLES:
|
|
64
64
|
${getExamples(preferredLanguage)}
|
|
65
65
|
|
|
66
66
|
Return JSON:
|
|
67
|
-
{
|
|
68
|
-
"description": "unique element identifier",
|
|
69
|
-
"error"?: "error message if any"
|
|
70
|
-
}`;
|
|
67
|
+
${describeElementReturnJsonSchema()}`;
|
|
71
68
|
};
|
|
72
69
|
export { elementDescriberInstruction };
|
|
73
70
|
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"ai-model/prompt/describe.mjs","sources":["../../../../src/ai-model/prompt/describe.ts"],"sourcesContent":["import { getPreferredLanguage } from '@midscene/shared/env';\n\nconst examplesMap: Record<string, string[]> = {\n Chinese: [\n '\"登录表单中的\"登录\"按钮\"',\n '\"搜索输入框,placeholder 为\"请输入关键词\"\"',\n '\"顶部导航栏中文字为\"首页\"的链接\"',\n '\"联系表单中的提交按钮\"',\n '\"aria-label 为\"打开菜单\"的菜单图标\"',\n '\"左侧导航栏中当前分组标题右侧的折叠图标\"',\n ],\n English: [\n '\"Login button with text \\'Sign In\\'\"',\n '\"Search input with placeholder \\'Enter keywords\\'\"',\n '\"Navigation link with text \\'Home\\' in header\"',\n '\"Submit button in contact form\"',\n '\"Menu icon with aria-label \\'Open menu\\'\"',\n '\"Collapse icon to the right of the current section title in the left sidebar\"',\n ],\n};\n\nconst getExamples = (language: string) => {\n const examples = examplesMap[language] || examplesMap.English;\n return examples.map((e) => `- ${e}`).join('\\n');\n};\n\nexport const elementDescriberInstruction = () => {\n const preferredLanguage = getPreferredLanguage();\n\n return `\nDescribe the element
|
|
1
|
+
{"version":3,"file":"ai-model/prompt/describe.mjs","sources":["../../../../src/ai-model/prompt/describe.ts"],"sourcesContent":["import { getPreferredLanguage } from '@midscene/shared/env';\n\nconst examplesMap: Record<string, string[]> = {\n Chinese: [\n '\"登录表单中的\"登录\"按钮\"',\n '\"搜索输入框,placeholder 为\"请输入关键词\"\"',\n '\"顶部导航栏中文字为\"首页\"的链接\"',\n '\"联系表单中的提交按钮\"',\n '\"aria-label 为\"打开菜单\"的菜单图标\"',\n '\"左侧导航栏中当前分组标题右侧的折叠图标\"',\n ],\n English: [\n '\"Login button with text \\'Sign In\\'\"',\n '\"Search input with placeholder \\'Enter keywords\\'\"',\n '\"Navigation link with text \\'Home\\' in header\"',\n '\"Submit button in contact form\"',\n '\"Menu icon with aria-label \\'Open menu\\'\"',\n '\"Collapse icon to the right of the current section title in the left sidebar\"',\n ],\n};\n\nconst getExamples = (language: string) => {\n const examples = examplesMap[language] || examplesMap.English;\n return examples.map((e) => `- ${e}`).join('\\n');\n};\n\nconst describeElementReturnJsonSchema = () => `{\n \"description\": \"unique element identifier\",\n \"error\"?: \"error message if any\"\n}`;\n\nexport const elementDescriberInstruction = () => {\n const preferredLanguage = getPreferredLanguage();\n\n return `\nDescribe the real page element indicated by the temporary callout.\nThe callout is an annotation overlay. It is not part of the page or target.\nThe description will be used later to locate the same element on the original screenshot without annotations, so write a locator-style description instead of a visual caption.\n\nIMPORTANT: You MUST write the description in ${preferredLanguage}.\n\nOBSERVE IN THIS ORDER:\n1. Target first: identify the smallest real UI part at the callout endpoint/center: text, glyph, icon, arrow, input, dropdown/select, option, button, link, status, checkbox, radio, switch, tab, menu item, slider, image, control, or empty region.\n2. Primitive: name what that smallest part is before adding surrounding context.\n3. Owner/context: add the nearest stable owner only when it helps disambiguate, such as a label, row/card title, column header, field name, or adjacent visible text.\n4. Similar candidates: if multiple candidates look similar, add stable local anchors from the same row, card, field, header, or group. Prefer visible text and values over inferred row counting or temporary visual state.\n\nRULES:\n- Keep description under 35 words.\n- Describe the smallest indicated UI part itself, not the larger container, row, card, sentence, or group that merely contains it.\n- Ignore every annotation overlay, including callout number, line, color, marker, border, dot, ring, crosshair, or selection box. Never describe the annotation as the target.\n- Do not borrow the text, glyph, direction, purpose, or state from a nearby element outside the callout endpoint/center.\n- For tiny or icon-only controls, name the visible glyph/control and add its owner/context; adjacent text is context, not the target. If similar tiny controls are adjacent in the same group, add local order or relative position inside that group.\n- If the endpoint/center is on a field value, label, or input body, describe that value/field/control. Do not retarget to a trailing icon, dropdown arrow, clear button, or search affordance unless the endpoint/center is on that icon itself.\n- If the endpoint/center is inside the bordered body, current value, trigger, or blank area of a select/dropdown/combobox/filter field, use primitive \"dropdown\" and describe that dropdown/select control. Do not call it an input unless it is clearly a free-text field.\n- If the endpoint/center is inside the bordered body or blank area of an input or filter field, describe the field body/current value/control even when the visible text is not exactly under the endpoint. Use the field label or visible value as context; do not snap to trailing icons or nearby table headers.\n- If the endpoint/center is on an expanded dropdown/select/menu list item, use primitive \"option\" for selectable list options or \"menuitem\" for command menu entries.\n- Only use primitive \"icon\" or \"arrow\" when the endpoint/center directly overlaps the real glyph strokes. A nearby search icon, dropdown arrow, clear button, or wrong locator result must not become the target primitive.\n- For compound controls or stacked glyphs, describe only the sub-part containing the callout endpoint/center, using upper/lower or left/right only when visible.\n- For inline text, links, or substrings, describe only the exact substring/link at the endpoint/center, not the whole line.\n- For repeated rows, cards, or options, use same-local anchors that are visible in the screenshot, such as neighboring cell text, field value, title, date, time, ID, or column/header label.\n- For tables or grids, describe the target as the intersection of the target column/header and same-row anchors. Do not use row ordinals or column ordinals unless the index/header is clearly visible.\n- Use selected, highlighted, hovered, focused, or active state only if the callout endpoint/center is inside that state.\n- If the endpoint/center is on blank space, describe the empty region/gap between stable surrounding anchors. Do not invent a nearby control.\n- Use actual visible text from the current screenshot when available; do not copy labels from the examples.\n- **Write the description in ${preferredLanguage}**\n\nEXAMPLES:\n${getExamples(preferredLanguage)}\n\nReturn JSON:\n${describeElementReturnJsonSchema()}`;\n};\n"],"names":["examplesMap","getExamples","language","examples","e","describeElementReturnJsonSchema","elementDescriberInstruction","preferredLanguage","getPreferredLanguage"],"mappings":";AAEA,MAAMA,cAAwC;IAC5C,SAAS;QACP;QACA;QACA;QACA;QACA;QACA;KACD;IACD,SAAS;QACP;QACA;QACA;QACA;QACA;QACA;KACD;AACH;AAEA,MAAMC,cAAc,CAACC;IACnB,MAAMC,WAAWH,WAAW,CAACE,SAAS,IAAIF,YAAY,OAAO;IAC7D,OAAOG,SAAS,GAAG,CAAC,CAACC,IAAM,CAAC,EAAE,EAAEA,GAAG,EAAE,IAAI,CAAC;AAC5C;AAEA,MAAMC,kCAAkC,IAAM,CAAC;;;CAG9C,CAAC;AAEK,MAAMC,8BAA8B;IACzC,MAAMC,oBAAoBC;IAE1B,OAAO,CAAC;;;;;6CAKmC,EAAED,kBAAkB;;;;;;;;;;;;;;;;;;;;;;;;;;6BA0BpC,EAAEA,kBAAkB;;;AAGjD,EAAEN,YAAYM,mBAAmB;;;AAGjC,EAAEF,mCAAmC;AACrC"}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { getPreferredLanguage } from "@midscene/shared/env";
|
|
2
|
-
import {
|
|
2
|
+
import { parseModelResponseJson } from "../service-caller/json.mjs";
|
|
3
3
|
import { extractXMLTag } from "./util.mjs";
|
|
4
4
|
function buildTypeQueryDemandValue(type, demand) {
|
|
5
5
|
const currentScreenshotConstraint = 'based on the current screenshot and its contents if provided, unless the user explicitly asks to compare with reference images';
|
|
@@ -8,19 +8,25 @@ function buildTypeQueryDemandValue(type, demand) {
|
|
|
8
8
|
return `${type}, ${currentScreenshotConstraint}, ${demand}`;
|
|
9
9
|
}
|
|
10
10
|
function parseXMLExtractionResponse(xmlString) {
|
|
11
|
-
const thought = extractXMLTag(xmlString, '
|
|
11
|
+
const thought = extractXMLTag(xmlString, 'observation');
|
|
12
12
|
const dataJsonStr = extractXMLTag(xmlString, 'data-json');
|
|
13
13
|
const errorsStr = extractXMLTag(xmlString, 'errors');
|
|
14
14
|
if (!dataJsonStr) throw new Error('Missing required field: data-json');
|
|
15
15
|
let data;
|
|
16
16
|
try {
|
|
17
|
-
data =
|
|
17
|
+
data = parseModelResponseJson(dataJsonStr, {
|
|
18
|
+
source: 'generic-object',
|
|
19
|
+
requireObject: false
|
|
20
|
+
});
|
|
18
21
|
} catch (e) {
|
|
19
22
|
throw new Error(`Failed to parse data-json: ${e}`);
|
|
20
23
|
}
|
|
21
24
|
let errors;
|
|
22
25
|
if (errorsStr) try {
|
|
23
|
-
const parsedErrors =
|
|
26
|
+
const parsedErrors = parseModelResponseJson(errorsStr, {
|
|
27
|
+
source: 'generic-object',
|
|
28
|
+
requireObject: false
|
|
29
|
+
});
|
|
24
30
|
if (Array.isArray(parsedErrors)) errors = parsedErrors;
|
|
25
31
|
} catch (e) {}
|
|
26
32
|
return {
|
|
@@ -58,7 +64,7 @@ When DATA_DEMAND is a JSON object, the keys in your response must exactly match
|
|
|
58
64
|
|
|
59
65
|
|
|
60
66
|
Return in the following XML format:
|
|
61
|
-
<
|
|
67
|
+
<observation>brief evidence observed for the extraction, less than 300 words. Use ${preferredLanguage} in this field.</observation>
|
|
62
68
|
<data-json>the extracted data as JSON. Make sure both the value and scheme meet the DATA_DEMAND. If you want to write some description in this field, use the same language as the DATA_DEMAND.</data-json>
|
|
63
69
|
<errors>optional error messages as JSON array, e.g., ["error1", "error2"]</errors>
|
|
64
70
|
|
|
@@ -75,7 +81,7 @@ For example, if the DATA_DEMAND is:
|
|
|
75
81
|
|
|
76
82
|
By viewing the screenshot and page contents, you can extract the following data:
|
|
77
83
|
|
|
78
|
-
<
|
|
84
|
+
<observation>According to the screenshot, i can see ...</observation>
|
|
79
85
|
<data-json>
|
|
80
86
|
{
|
|
81
87
|
"name": "John",
|
|
@@ -93,7 +99,7 @@ the todo items list, string[]
|
|
|
93
99
|
|
|
94
100
|
By viewing the screenshot and page contents, you can extract the following data:
|
|
95
101
|
|
|
96
|
-
<
|
|
102
|
+
<observation>According to the screenshot, i can see ...</observation>
|
|
97
103
|
<data-json>
|
|
98
104
|
["todo 1", "todo 2", "todo 3"]
|
|
99
105
|
</data-json>
|
|
@@ -107,7 +113,7 @@ the page title, string
|
|
|
107
113
|
|
|
108
114
|
By viewing the screenshot and page contents, you can extract the following data:
|
|
109
115
|
|
|
110
|
-
<
|
|
116
|
+
<observation>According to the screenshot, i can see ...</observation>
|
|
111
117
|
<data-json>
|
|
112
118
|
"todo list"
|
|
113
119
|
</data-json>
|
|
@@ -123,7 +129,7 @@ If the DATA_DEMAND is:
|
|
|
123
129
|
|
|
124
130
|
By viewing the screenshot and page contents, you can extract the following data:
|
|
125
131
|
|
|
126
|
-
<
|
|
132
|
+
<observation>According to the screenshot, i can see ...</observation>
|
|
127
133
|
<data-json>
|
|
128
134
|
{ "StatementIsTruthy": true }
|
|
129
135
|
</data-json>
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"ai-model/prompt/extraction.mjs","sources":["../../../../src/ai-model/prompt/extraction.ts"],"sourcesContent":["import type { AIDataExtractionResponse, ServiceExtractParam } from '@/types';\nimport { getPreferredLanguage } from '@midscene/shared/env';\nimport {
|
|
1
|
+
{"version":3,"file":"ai-model/prompt/extraction.mjs","sources":["../../../../src/ai-model/prompt/extraction.ts"],"sourcesContent":["import type { AIDataExtractionResponse, ServiceExtractParam } from '@/types';\nimport { getPreferredLanguage } from '@midscene/shared/env';\nimport { parseModelResponseJson } from '../service-caller/json';\nimport { extractXMLTag } from './util';\n\nexport function buildTypeQueryDemandValue(\n type: 'Boolean' | 'Number' | 'String' | 'Assert' | 'WaitFor',\n demand: ServiceExtractParam,\n) {\n const currentScreenshotConstraint =\n 'based on the current screenshot and its contents if provided, unless the user explicitly asks to compare with reference images';\n\n if (type === 'Assert') {\n return `Boolean, ${currentScreenshotConstraint}, whether the following statement is true: ${demand}`;\n }\n\n if (type === 'WaitFor') {\n return `Boolean, the user wants to do some 'wait for' operation. ${currentScreenshotConstraint}, please check whether the following statement is true: ${demand}`;\n }\n\n return `${type}, ${currentScreenshotConstraint}, ${demand}`;\n}\n\n/**\n * Parse XML response from LLM and convert to AIDataExtractionResponse\n */\nexport function parseXMLExtractionResponse<T>(\n xmlString: string,\n): AIDataExtractionResponse<T> {\n // Keep the internal field named `thought`, but ask models to emit\n // <observation>. Gemini may only return <thought>-named content when\n // thinking summaries are enabled.\n const thought = extractXMLTag(xmlString, 'observation');\n const dataJsonStr = extractXMLTag(xmlString, 'data-json');\n const errorsStr = extractXMLTag(xmlString, 'errors');\n\n // Parse data-json (required)\n if (!dataJsonStr) {\n throw new Error('Missing required field: data-json');\n }\n\n let data: T;\n try {\n data = parseModelResponseJson(dataJsonStr, {\n source: 'generic-object',\n requireObject: false,\n }) as T;\n } catch (e) {\n throw new Error(`Failed to parse data-json: ${e}`);\n }\n\n // Parse errors (optional)\n let errors: string[] | undefined;\n if (errorsStr) {\n try {\n const parsedErrors = parseModelResponseJson(errorsStr, {\n source: 'generic-object',\n requireObject: false,\n });\n if (Array.isArray(parsedErrors)) {\n errors = parsedErrors;\n }\n } catch (e) {\n // If errors parsing fails, just ignore it\n }\n }\n\n return {\n ...(thought ? { thought } : {}),\n data,\n ...(errors && errors.length > 0 ? { errors } : {}),\n };\n}\n\nexport function systemPromptToExtract(options?: {\n screenshotIncluded?: boolean;\n referenceImagesIncluded?: boolean;\n}) {\n const preferredLanguage = getPreferredLanguage();\n const screenshotIncluded = options?.screenshotIncluded ?? true;\n const referenceImagesIncluded = options?.referenceImagesIncluded ?? false;\n\n const contextPrompts = [\n \"The user will give you data requirements in <DATA_DEMAND>. You need to understand the user's requirements and extract the data satisfying the <DATA_DEMAND>.\",\n ];\n\n if (screenshotIncluded) {\n contextPrompts.push(\n 'The user will provide a current screenshot to evaluate, and may provide its contents. Base your answer on the current screenshot and its contents when provided. Treat them as the primary source of truth for what is currently visible or true.',\n );\n } else {\n contextPrompts.push(\n 'The user will not provide a current screenshot. Use only the supplied page contents and other inputs, and do not infer unsupported visual details.',\n );\n }\n\n if (referenceImagesIncluded) {\n const referenceImagesPrompt =\n 'Reference images are supporting context only unless <DATA_DEMAND> explicitly asks for comparison, matching, or reasoning about them.';\n contextPrompts.push(\n screenshotIncluded\n ? `${referenceImagesPrompt} Do not conclude that something exists in the current screenshot solely because it appears in a reference image; when they conflict, trust the current screenshot and its contents.`\n : `${referenceImagesPrompt} Do not treat reference images as direct evidence of the current state unless the demand explicitly asks you to use them that way.`,\n );\n }\n const contextPrompt = contextPrompts.join('\\n\\n');\n\n return `\nYou are a versatile professional in software UI design and testing. Your outstanding contributions will impact the user experience of billions of users.\n\n${contextPrompt}\n\nIf a key specifies a JSON data type (such as Number, String, Boolean, Object, Array), ensure the returned value strictly matches that data type.\n\nWhen DATA_DEMAND is a JSON object, the keys in your response must exactly match the keys in DATA_DEMAND. Do not rename, translate, or substitute any key.\n\n\nReturn in the following XML format:\n<observation>brief evidence observed for the extraction, less than 300 words. Use ${preferredLanguage} in this field.</observation>\n<data-json>the extracted data as JSON. Make sure both the value and scheme meet the DATA_DEMAND. If you want to write some description in this field, use the same language as the DATA_DEMAND.</data-json>\n<errors>optional error messages as JSON array, e.g., [\"error1\", \"error2\"]</errors>\n\n# Example 1\nFor example, if the DATA_DEMAND is:\n\n<DATA_DEMAND>\n{\n \"name\": \"name shows on the left panel, string\",\n \"age\": \"age shows on the right panel, number\",\n \"isAdmin\": \"if the user is admin, boolean\"\n}\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<observation>According to the screenshot, i can see ...</observation>\n<data-json>\n{\n \"name\": \"John\",\n \"age\": 30,\n \"isAdmin\": true\n}\n</data-json>\n\n# Example 2\nIf the DATA_DEMAND is:\n\n<DATA_DEMAND>\nthe todo items list, string[]\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<observation>According to the screenshot, i can see ...</observation>\n<data-json>\n[\"todo 1\", \"todo 2\", \"todo 3\"]\n</data-json>\n\n# Example 3\nIf the DATA_DEMAND is:\n\n<DATA_DEMAND>\nthe page title, string\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<observation>According to the screenshot, i can see ...</observation>\n<data-json>\n\"todo list\"\n</data-json>\n\n# Example 4\nIf the DATA_DEMAND is:\n\n<DATA_DEMAND>\n{\n \"StatementIsTruthy\": \"Boolean, is it currently the SMS page?\"\n}\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<observation>According to the screenshot, i can see ...</observation>\n<data-json>\n{ \"StatementIsTruthy\": true }\n</data-json>\n`;\n}\n\nexport const extractDataQueryPrompt = (\n pageDescription: string,\n dataQuery: string | Record<string, string>,\n) => {\n let dataQueryText = '';\n if (typeof dataQuery === 'string') {\n dataQueryText = dataQuery;\n } else {\n dataQueryText = JSON.stringify(dataQuery, null, 2);\n }\n\n return `\n<PageDescription>\n${pageDescription}\n</PageDescription>\n\n<DATA_DEMAND>\n${dataQueryText}\n</DATA_DEMAND>\n `;\n};\n"],"names":["buildTypeQueryDemandValue","type","demand","currentScreenshotConstraint","parseXMLExtractionResponse","xmlString","thought","extractXMLTag","dataJsonStr","errorsStr","Error","data","parseModelResponseJson","e","errors","parsedErrors","Array","systemPromptToExtract","options","preferredLanguage","getPreferredLanguage","screenshotIncluded","referenceImagesIncluded","contextPrompts","referenceImagesPrompt","contextPrompt","extractDataQueryPrompt","pageDescription","dataQuery","dataQueryText","JSON"],"mappings":";;;AAKO,SAASA,0BACdC,IAA4D,EAC5DC,MAA2B;IAE3B,MAAMC,8BACJ;IAEF,IAAIF,AAAS,aAATA,MACF,OAAO,CAAC,SAAS,EAAEE,4BAA4B,2CAA2C,EAAED,QAAQ;IAGtG,IAAID,AAAS,cAATA,MACF,OAAO,CAAC,yDAAyD,EAAEE,4BAA4B,wDAAwD,EAAED,QAAQ;IAGnK,OAAO,GAAGD,KAAK,EAAE,EAAEE,4BAA4B,EAAE,EAAED,QAAQ;AAC7D;AAKO,SAASE,2BACdC,SAAiB;IAKjB,MAAMC,UAAUC,cAAcF,WAAW;IACzC,MAAMG,cAAcD,cAAcF,WAAW;IAC7C,MAAMI,YAAYF,cAAcF,WAAW;IAG3C,IAAI,CAACG,aACH,MAAM,IAAIE,MAAM;IAGlB,IAAIC;IACJ,IAAI;QACFA,OAAOC,uBAAuBJ,aAAa;YACzC,QAAQ;YACR,eAAe;QACjB;IACF,EAAE,OAAOK,GAAG;QACV,MAAM,IAAIH,MAAM,CAAC,2BAA2B,EAAEG,GAAG;IACnD;IAGA,IAAIC;IACJ,IAAIL,WACF,IAAI;QACF,MAAMM,eAAeH,uBAAuBH,WAAW;YACrD,QAAQ;YACR,eAAe;QACjB;QACA,IAAIO,MAAM,OAAO,CAACD,eAChBD,SAASC;IAEb,EAAE,OAAOF,GAAG,CAEZ;IAGF,OAAO;QACL,GAAIP,UAAU;YAAEA;QAAQ,IAAI,CAAC,CAAC;QAC9BK;QACA,GAAIG,UAAUA,OAAO,MAAM,GAAG,IAAI;YAAEA;QAAO,IAAI,CAAC,CAAC;IACnD;AACF;AAEO,SAASG,sBAAsBC,OAGrC;IACC,MAAMC,oBAAoBC;IAC1B,MAAMC,qBAAqBH,SAAS,sBAAsB;IAC1D,MAAMI,0BAA0BJ,SAAS,2BAA2B;IAEpE,MAAMK,iBAAiB;QACrB;KACD;IAED,IAAIF,oBACFE,eAAe,IAAI,CACjB;SAGFA,eAAe,IAAI,CACjB;IAIJ,IAAID,yBAAyB;QAC3B,MAAME,wBACJ;QACFD,eAAe,IAAI,CACjBF,qBACI,GAAGG,sBAAsB,mLAAmL,CAAC,GAC7M,GAAGA,sBAAsB,kIAAkI,CAAC;IAEpK;IACA,MAAMC,gBAAgBF,eAAe,IAAI,CAAC;IAE1C,OAAO,CAAC;;;AAGV,EAAEE,cAAc;;;;;;;;kFAQkE,EAAEN,kBAAkB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEtG,CAAC;AACD;AAEO,MAAMO,yBAAyB,CACpCC,iBACAC;IAEA,IAAIC,gBAAgB;IAElBA,gBADE,AAAqB,YAArB,OAAOD,YACOA,YAEAE,KAAK,SAAS,CAACF,WAAW,MAAM;IAGlD,OAAO,CAAC;;AAEV,EAAED,gBAAgB;;;;AAIlB,EAAEE,cAAc;;EAEd,CAAC;AACH"}
|
|
@@ -112,16 +112,16 @@ async function systemPromptToTaskPlanning({ actionSpace, locatePromptSpec, inclu
|
|
|
112
112
|
const locateExample1 = locateExample('Add to cart button for Sauce Labs Backpack', 1);
|
|
113
113
|
const locateNameField = locateExample('Name input field in the registration form', 2);
|
|
114
114
|
const locateEmailField = locateExample('Email input field in the registration form', 3);
|
|
115
|
-
const step1Title = shouldIncludeSubGoals ? '## Step 1: Observe and Plan (related tags: <
|
|
115
|
+
const step1Title = shouldIncludeSubGoals ? '## Step 1: Observe and Plan (related tags: <planning>, <update-plan-content>, <mark-sub-goal-done>)' : '## Step 1: Observe (related tags: <planning>)';
|
|
116
116
|
const step1Description = shouldIncludeSubGoals ? "First, observe the current screenshot and previous logs, then break down the user's instruction into multiple high-level sub-goals. Update the status of sub-goals based on what you see in the current screenshot." : 'First, observe the current screenshot and previous logs to understand the current state.';
|
|
117
117
|
const explicitInstructionRule = 'CRITICAL - Following Explicit Instructions: When the user gives you specific operation steps (not high-level goals), you MUST execute ONLY those exact steps - nothing more, nothing less. Do NOT add extra actions even if they seem logical. For example: "fill out the form" means only fill fields, do NOT submit; "click the button" means only click, do NOT wait for page load or verify results; "type \'hello\'" means only type, do NOT press Enter.';
|
|
118
|
-
const
|
|
118
|
+
const planningTagDescription = shouldIncludeSubGoals ? `REQUIRED: You MUST always output the <planning> tag. Never skip it.
|
|
119
119
|
|
|
120
|
-
Include your
|
|
120
|
+
Include your planning details in the <planning> tag. It should answer: What is the user's requirement? What is the current state based on the screenshot? Are all sub-goals completed? If not, what should be the next action? Write it naturally without numbering or section headers.
|
|
121
121
|
|
|
122
|
-
${explicitInstructionRule}` : `REQUIRED: You MUST always output the <
|
|
122
|
+
${explicitInstructionRule}` : `REQUIRED: You MUST always output the <planning> tag. Never skip it.
|
|
123
123
|
|
|
124
|
-
Include your
|
|
124
|
+
Include your planning details in the <planning> tag. It should answer: What is the current state based on the screenshot? What should be the next action? Write it naturally without numbering or section headers.
|
|
125
125
|
|
|
126
126
|
${explicitInstructionRule}`;
|
|
127
127
|
const subGoalTags = shouldIncludeSubGoals ? `
|
|
@@ -154,7 +154,7 @@ During execution, you can call <update-plan-content> at any time to update the p
|
|
|
154
154
|
|
|
155
155
|
If the user wants to "log in to a system using username and password, complete all to-do items, and submit a registration form", you can break it down into the following sub-goals:
|
|
156
156
|
|
|
157
|
-
<
|
|
157
|
+
<planning>...</planning>
|
|
158
158
|
<update-plan-content>
|
|
159
159
|
<sub-goal index="1" status="pending">Log in to the system</sub-goal>
|
|
160
160
|
<sub-goal index="2" status="pending">Complete all to-do items</sub-goal>
|
|
@@ -190,9 +190,9 @@ ${step1Title}
|
|
|
190
190
|
|
|
191
191
|
${step1Description}
|
|
192
192
|
${shouldIncludeSubGoals ? `\n${OBSERVE_STEP_NOTES}\n` : ''}
|
|
193
|
-
* <
|
|
193
|
+
* <planning> tag (REQUIRED)
|
|
194
194
|
|
|
195
|
-
${
|
|
195
|
+
${planningTagDescription}
|
|
196
196
|
${subGoalTags}
|
|
197
197
|
${shouldIncludeSubGoals ? `
|
|
198
198
|
## Step ${memoryStepNumber}: Memory Data from Current Screenshot (related tags: <memory>)
|
|
@@ -335,7 +335,7 @@ Return in XML format following this decision flow:
|
|
|
335
335
|
|
|
336
336
|
**Always include (REQUIRED):**
|
|
337
337
|
<!-- Step 1: Observe${shouldIncludeSubGoals ? ' and Plan' : ''} -->
|
|
338
|
-
<
|
|
338
|
+
<planning>Your planning details here. NEVER skip this tag.</planning>
|
|
339
339
|
${shouldIncludeSubGoals ? `
|
|
340
340
|
<!-- required when no update-plan-content is provided in the previous response -->
|
|
341
341
|
<update-plan-content>...</update-plan-content>
|
|
@@ -374,7 +374,7 @@ Below is an example of a multi-turn conversation for "fill out the registration
|
|
|
374
374
|
**Screenshot:** [Shows a registration form with empty Name and Email fields]
|
|
375
375
|
|
|
376
376
|
**Your response:**
|
|
377
|
-
<
|
|
377
|
+
<planning>The user wants me to fill out the registration form with specific values and return the email address. I can see the form has two fields: Name and Email. Both are currently empty. I'll break this down into sub-goals and start with the Name field. Note: The instruction is to fill the form only (not submit), and return the email at the end.</planning>
|
|
378
378
|
<update-plan-content>
|
|
379
379
|
<sub-goal index="1" status="pending">Fill in the Name field with 'John'</sub-goal>
|
|
380
380
|
<sub-goal index="2" status="pending">Fill in the Email field with 'john@example.com'</sub-goal>
|
|
@@ -404,7 +404,7 @@ Actions performed for current sub-goal:
|
|
|
404
404
|
**Screenshot:** [Shows the form with Name field now focused/active]
|
|
405
405
|
|
|
406
406
|
**Your response:**
|
|
407
|
-
<
|
|
407
|
+
<planning>The Name field is now focused. I need to type 'John' into this field. Current sub-goal is running, will be completed after input.</planning>
|
|
408
408
|
<log>Typing 'John' into the Name field</log>
|
|
409
409
|
<action-type>Input</action-type>
|
|
410
410
|
<action-param-json>
|
|
@@ -430,7 +430,7 @@ Actions performed for current sub-goal:
|
|
|
430
430
|
**Screenshot:** [Shows the form with Name field containing 'John']
|
|
431
431
|
|
|
432
432
|
**Your response:**
|
|
433
|
-
<
|
|
433
|
+
<planning>The Name field now shows 'John'. Sub-goal 1 is complete. Next, I need to fill the Email field for sub-goal 2.</planning>
|
|
434
434
|
<mark-sub-goal-done>
|
|
435
435
|
<sub-goal index="1" status="finished" />
|
|
436
436
|
</mark-sub-goal-done>
|
|
@@ -459,7 +459,7 @@ Actions performed for current sub-goal:
|
|
|
459
459
|
**Screenshot:** [Shows the form with Name='John' and Email field focused]
|
|
460
460
|
|
|
461
461
|
**Your response:**
|
|
462
|
-
<
|
|
462
|
+
<planning>The Email field is now focused. I'll enter 'john@example.com'. After this, sub-goal 2 will be complete and my task will be done.</planning>
|
|
463
463
|
<log>Typing email address into the Email field</log>
|
|
464
464
|
<action-type>Input</action-type>
|
|
465
465
|
<action-param-json>
|
|
@@ -485,7 +485,7 @@ Actions performed for current sub-goal:
|
|
|
485
485
|
**Screenshot:** [Shows the form with Name='John' and Email='john@example.com']
|
|
486
486
|
|
|
487
487
|
**Your response:**
|
|
488
|
-
<
|
|
488
|
+
<planning>Both fields are now filled: Name shows 'John' and Email shows 'john@example.com'. Sub-goal 2 is complete. The user asked me to return the filled email address, so I need to include 'john@example.com' in my response. All sub-goals are now finished.</planning>
|
|
489
489
|
<mark-sub-goal-done>
|
|
490
490
|
<sub-goal index="2" status="finished" />
|
|
491
491
|
<sub-goal index="3" status="finished" />
|
|
@@ -504,7 +504,7 @@ Below is an example of a multi-turn conversation for "fill out the registration
|
|
|
504
504
|
**Screenshot:** [Shows a registration form with empty Name and Email fields]
|
|
505
505
|
|
|
506
506
|
**Your response:**
|
|
507
|
-
<
|
|
507
|
+
<planning>The user wants me to fill out the registration form with specific values and return the email address. I can see the form has two fields: Name and Email. Both are currently empty. I should start by clicking on the Name field. Note: The instruction is to fill the form only (not submit), and return the email at the end.</planning>
|
|
508
508
|
<log>Click on the Name field to start filling the form</log>
|
|
509
509
|
<action-type>Tap</action-type>
|
|
510
510
|
<action-param-json>
|
|
@@ -521,7 +521,7 @@ The previous action has been executed, here is the latest screenshot. Please con
|
|
|
521
521
|
**Screenshot:** [Shows the form with Name field now focused/active]
|
|
522
522
|
|
|
523
523
|
**Your response:**
|
|
524
|
-
<
|
|
524
|
+
<planning>The Name field is now focused. I need to type 'John' into this field. After this, I'll need to fill the Email field as well.</planning>
|
|
525
525
|
<log>Typing 'John' into the Name field</log>
|
|
526
526
|
<action-type>Input</action-type>
|
|
527
527
|
<action-param-json>
|
|
@@ -538,7 +538,7 @@ The previous action has been executed, here is the latest screenshot. Please con
|
|
|
538
538
|
**Screenshot:** [Shows the form with Name field containing 'John']
|
|
539
539
|
|
|
540
540
|
**Your response:**
|
|
541
|
-
<
|
|
541
|
+
<planning>The Name field now shows 'John'. Next, I need to fill the Email field with 'john@example.com'. I'll click on the Email field first.</planning>
|
|
542
542
|
<log>Moving to the Email field</log>
|
|
543
543
|
<action-type>Tap</action-type>
|
|
544
544
|
<action-param-json>
|
|
@@ -555,7 +555,7 @@ The previous action has been executed, here is the latest screenshot. Please con
|
|
|
555
555
|
**Screenshot:** [Shows the form with Name='John' and Email field focused]
|
|
556
556
|
|
|
557
557
|
**Your response:**
|
|
558
|
-
<
|
|
558
|
+
<planning>The Email field is now focused. I'll enter 'john@example.com'. After this, both required fields will be filled and my task will be complete.</planning>
|
|
559
559
|
<log>Typing email address into the Email field</log>
|
|
560
560
|
<action-type>Input</action-type>
|
|
561
561
|
<action-param-json>
|
|
@@ -572,7 +572,7 @@ The previous action has been executed, here is the latest screenshot. Please con
|
|
|
572
572
|
**Screenshot:** [Shows the form with Name='John' and Email='john@example.com']
|
|
573
573
|
|
|
574
574
|
**Your response:**
|
|
575
|
-
<
|
|
575
|
+
<planning>Both fields are now filled: Name shows 'John' and Email shows 'john@example.com'. The user asked me to return the filled email address, so I should include 'john@example.com' in my response. The instruction has been fulfilled.</planning>
|
|
576
576
|
<complete success="true">john@example.com</complete>
|
|
577
577
|
`}`;
|
|
578
578
|
}
|