@midscene/core 1.10.2-beta-20260706032158.0 → 1.10.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (161) hide show
  1. package/dist/es/agent/agent.mjs +40 -1
  2. package/dist/es/agent/agent.mjs.map +1 -1
  3. package/dist/es/agent/metrics.mjs +81 -0
  4. package/dist/es/agent/metrics.mjs.map +1 -0
  5. package/dist/es/agent/task-builder.mjs +1 -1
  6. package/dist/es/agent/task-builder.mjs.map +1 -1
  7. package/dist/es/agent/utils.mjs +1 -1
  8. package/dist/es/ai-model/connectivity/run-connectivity-test.mjs +9 -3
  9. package/dist/es/ai-model/connectivity/run-connectivity-test.mjs.map +1 -1
  10. package/dist/es/ai-model/inspect.mjs +3 -3
  11. package/dist/es/ai-model/inspect.mjs.map +1 -1
  12. package/dist/es/ai-model/llm-planning.mjs +1 -1
  13. package/dist/es/ai-model/llm-planning.mjs.map +1 -1
  14. package/dist/es/ai-model/model-adapter/resolve.mjs +2 -2
  15. package/dist/es/ai-model/model-adapter/resolve.mjs.map +1 -1
  16. package/dist/es/ai-model/models/default.mjs +2 -1
  17. package/dist/es/ai-model/models/default.mjs.map +1 -1
  18. package/dist/es/ai-model/models/doubao.mjs +38 -60
  19. package/dist/es/ai-model/models/doubao.mjs.map +1 -1
  20. package/dist/es/ai-model/models/gemini.mjs +4 -15
  21. package/dist/es/ai-model/models/gemini.mjs.map +1 -1
  22. package/dist/es/ai-model/models/glm.mjs +4 -0
  23. package/dist/es/ai-model/models/glm.mjs.map +1 -1
  24. package/dist/es/ai-model/models/gpt.mjs +5 -1
  25. package/dist/es/ai-model/models/gpt.mjs.map +1 -1
  26. package/dist/es/ai-model/models/kimi.mjs +4 -0
  27. package/dist/es/ai-model/models/kimi.mjs.map +1 -1
  28. package/dist/es/ai-model/models/mimo.mjs +3 -2
  29. package/dist/es/ai-model/models/mimo.mjs.map +1 -1
  30. package/dist/es/ai-model/models/qwen.mjs.map +1 -1
  31. package/dist/es/ai-model/models/ui-tars/adapter.mjs +2 -41
  32. package/dist/es/ai-model/models/ui-tars/adapter.mjs.map +1 -1
  33. package/dist/es/ai-model/models/utils/intent.mjs +6 -0
  34. package/dist/es/ai-model/models/utils/intent.mjs.map +1 -0
  35. package/dist/es/ai-model/prompt/describe.mjs +31 -34
  36. package/dist/es/ai-model/prompt/describe.mjs.map +1 -1
  37. package/dist/es/ai-model/prompt/extraction.mjs +15 -9
  38. package/dist/es/ai-model/prompt/extraction.mjs.map +1 -1
  39. package/dist/es/ai-model/prompt/llm-planning.mjs +19 -19
  40. package/dist/es/ai-model/prompt/llm-planning.mjs.map +1 -1
  41. package/dist/es/ai-model/prompt/locate-grounding-rules.mjs +7 -3
  42. package/dist/es/ai-model/prompt/locate-grounding-rules.mjs.map +1 -1
  43. package/dist/es/ai-model/prompt/recorder-metadata-generator.mjs +2 -1
  44. package/dist/es/ai-model/prompt/recorder-metadata-generator.mjs.map +1 -1
  45. package/dist/es/ai-model/service-caller/index.mjs +40 -19
  46. package/dist/es/ai-model/service-caller/index.mjs.map +1 -1
  47. package/dist/es/ai-model/service-caller/json.mjs +31 -28
  48. package/dist/es/ai-model/service-caller/json.mjs.map +1 -1
  49. package/dist/es/device/index.mjs +2 -4
  50. package/dist/es/device/index.mjs.map +1 -1
  51. package/dist/es/dump/html-utils.mjs +63 -1
  52. package/dist/es/dump/html-utils.mjs.map +1 -1
  53. package/dist/es/dump/screenshot-restoration.mjs +6 -0
  54. package/dist/es/dump/screenshot-restoration.mjs.map +1 -1
  55. package/dist/es/element-describer.mjs +36 -18
  56. package/dist/es/element-describer.mjs.map +1 -1
  57. package/dist/es/index.mjs +2 -2
  58. package/dist/es/index.mjs.map +1 -1
  59. package/dist/es/report-generator.mjs +28 -1
  60. package/dist/es/report-generator.mjs.map +1 -1
  61. package/dist/es/report-markdown.mjs +300 -35
  62. package/dist/es/report-markdown.mjs.map +1 -1
  63. package/dist/es/report.mjs +29 -1
  64. package/dist/es/report.mjs.map +1 -1
  65. package/dist/es/service/index.mjs +101 -25
  66. package/dist/es/service/index.mjs.map +1 -1
  67. package/dist/es/service/utils.mjs +65 -1
  68. package/dist/es/service/utils.mjs.map +1 -1
  69. package/dist/es/types.mjs.map +1 -1
  70. package/dist/es/utils.mjs +2 -2
  71. package/dist/lib/agent/agent.js +40 -1
  72. package/dist/lib/agent/agent.js.map +1 -1
  73. package/dist/lib/agent/metrics.js +115 -0
  74. package/dist/lib/agent/metrics.js.map +1 -0
  75. package/dist/lib/agent/task-builder.js +1 -1
  76. package/dist/lib/agent/task-builder.js.map +1 -1
  77. package/dist/lib/agent/utils.js +1 -1
  78. package/dist/lib/ai-model/connectivity/run-connectivity-test.js +9 -3
  79. package/dist/lib/ai-model/connectivity/run-connectivity-test.js.map +1 -1
  80. package/dist/lib/ai-model/inspect.js +3 -3
  81. package/dist/lib/ai-model/inspect.js.map +1 -1
  82. package/dist/lib/ai-model/llm-planning.js +1 -1
  83. package/dist/lib/ai-model/llm-planning.js.map +1 -1
  84. package/dist/lib/ai-model/model-adapter/resolve.js +1 -1
  85. package/dist/lib/ai-model/model-adapter/resolve.js.map +1 -1
  86. package/dist/lib/ai-model/models/default.js +2 -1
  87. package/dist/lib/ai-model/models/default.js.map +1 -1
  88. package/dist/lib/ai-model/models/doubao.js +38 -69
  89. package/dist/lib/ai-model/models/doubao.js.map +1 -1
  90. package/dist/lib/ai-model/models/gemini.js +4 -15
  91. package/dist/lib/ai-model/models/gemini.js.map +1 -1
  92. package/dist/lib/ai-model/models/glm.js +4 -0
  93. package/dist/lib/ai-model/models/glm.js.map +1 -1
  94. package/dist/lib/ai-model/models/gpt.js +5 -1
  95. package/dist/lib/ai-model/models/gpt.js.map +1 -1
  96. package/dist/lib/ai-model/models/kimi.js +4 -0
  97. package/dist/lib/ai-model/models/kimi.js.map +1 -1
  98. package/dist/lib/ai-model/models/mimo.js +3 -2
  99. package/dist/lib/ai-model/models/mimo.js.map +1 -1
  100. package/dist/lib/ai-model/models/qwen.js.map +1 -1
  101. package/dist/lib/ai-model/models/ui-tars/adapter.js +1 -40
  102. package/dist/lib/ai-model/models/ui-tars/adapter.js.map +1 -1
  103. package/dist/lib/ai-model/models/utils/intent.js +40 -0
  104. package/dist/lib/ai-model/models/utils/intent.js.map +1 -0
  105. package/dist/lib/ai-model/prompt/describe.js +31 -34
  106. package/dist/lib/ai-model/prompt/describe.js.map +1 -1
  107. package/dist/lib/ai-model/prompt/extraction.js +14 -8
  108. package/dist/lib/ai-model/prompt/extraction.js.map +1 -1
  109. package/dist/lib/ai-model/prompt/llm-planning.js +19 -19
  110. package/dist/lib/ai-model/prompt/llm-planning.js.map +1 -1
  111. package/dist/lib/ai-model/prompt/locate-grounding-rules.js +7 -3
  112. package/dist/lib/ai-model/prompt/locate-grounding-rules.js.map +1 -1
  113. package/dist/lib/ai-model/prompt/recorder-metadata-generator.js +4 -3
  114. package/dist/lib/ai-model/prompt/recorder-metadata-generator.js.map +1 -1
  115. package/dist/lib/ai-model/service-caller/index.js +44 -23
  116. package/dist/lib/ai-model/service-caller/index.js.map +1 -1
  117. package/dist/lib/ai-model/service-caller/json.js +33 -33
  118. package/dist/lib/ai-model/service-caller/json.js.map +1 -1
  119. package/dist/lib/device/index.js +2 -4
  120. package/dist/lib/device/index.js.map +1 -1
  121. package/dist/lib/dump/html-utils.js +65 -0
  122. package/dist/lib/dump/html-utils.js.map +1 -1
  123. package/dist/lib/dump/screenshot-restoration.js +6 -0
  124. package/dist/lib/dump/screenshot-restoration.js.map +1 -1
  125. package/dist/lib/element-describer.js +37 -22
  126. package/dist/lib/element-describer.js.map +1 -1
  127. package/dist/lib/index.js +1 -4
  128. package/dist/lib/index.js.map +1 -1
  129. package/dist/lib/report-generator.js +27 -0
  130. package/dist/lib/report-generator.js.map +1 -1
  131. package/dist/lib/report-markdown.js +300 -35
  132. package/dist/lib/report-markdown.js.map +1 -1
  133. package/dist/lib/report.js +28 -0
  134. package/dist/lib/report.js.map +1 -1
  135. package/dist/lib/service/index.js +99 -23
  136. package/dist/lib/service/index.js.map +1 -1
  137. package/dist/lib/service/utils.js +98 -1
  138. package/dist/lib/service/utils.js.map +1 -1
  139. package/dist/lib/types.js.map +1 -1
  140. package/dist/lib/utils.js +2 -2
  141. package/dist/types/agent/agent.d.ts +15 -2
  142. package/dist/types/agent/index.d.ts +1 -0
  143. package/dist/types/agent/metrics.d.ts +48 -0
  144. package/dist/types/ai-model/model-adapter/types.d.ts +8 -0
  145. package/dist/types/ai-model/models/doubao.d.ts +1 -4
  146. package/dist/types/ai-model/models/gemini.d.ts +1 -1
  147. package/dist/types/ai-model/models/registry.d.ts +1 -1
  148. package/dist/types/ai-model/models/utils/intent.d.ts +2 -0
  149. package/dist/types/ai-model/service-caller/index.d.ts +8 -3
  150. package/dist/types/ai-model/service-caller/json.d.ts +19 -2
  151. package/dist/types/device/device-options.d.ts +0 -28
  152. package/dist/types/device/index.d.ts +0 -5
  153. package/dist/types/dump/html-utils.d.ts +2 -0
  154. package/dist/types/dump/screenshot-restoration.d.ts +1 -1
  155. package/dist/types/element-describer.d.ts +3 -6
  156. package/dist/types/index.d.ts +2 -2
  157. package/dist/types/report-generator.d.ts +6 -0
  158. package/dist/types/service/index.d.ts +1 -1
  159. package/dist/types/service/utils.d.ts +21 -2
  160. package/dist/types/types.d.ts +12 -2
  161. package/package.json +3 -3
@@ -21,53 +21,50 @@ const getExamples = (language)=>{
21
21
  const examples = examplesMap[language] || examplesMap.English;
22
22
  return examples.map((e)=>`- ${e}`).join('\n');
23
23
  };
24
+ const describeElementReturnJsonSchema = ()=>`{
25
+ "description": "unique element identifier",
26
+ "error"?: "error message if any"
27
+ }`;
24
28
  const elementDescriberInstruction = ()=>{
25
29
  const preferredLanguage = getPreferredLanguage();
26
30
  return `
27
- Describe the element in the red rectangle for precise identification.
31
+ Describe the real page element indicated by the temporary callout.
32
+ The callout is an annotation overlay. It is not part of the page or target.
33
+ The description will be used later to locate the same element on the original screenshot without annotations, so write a locator-style description instead of a visual caption.
28
34
 
29
35
  IMPORTANT: You MUST write the description in ${preferredLanguage}.
30
36
 
31
- CRITICAL REQUIREMENTS:
32
- 1. UNIQUENESS: The description must uniquely identify this element on the current page
33
- 2. UNIVERSALITY: Use generic, reusable selectors that work across different contexts
34
- 3. PRECISION: Be specific enough to distinguish from similar elements
35
-
36
- DESCRIPTION STRUCTURE:
37
- 1. Element type (button, input, link, div, etc.)
38
- 2. Primary identifier (in order of preference):
39
- - Unique text content: "with text 'Login'"
40
- - Unique attribute: "with aria-label 'Search'"
41
- - Unique class/ID: "with class 'primary-button'"
42
- - Unique position: "in header navigation"
43
- 3. Secondary identifiers (if needed for uniqueness):
44
- - Visual features: "blue background", "with icon"
45
- - Relative position: "below search bar", "in sidebar"
46
- - Parent context: "in login form", "in main menu"
47
- - Neighboring stable text: "to the right of the 'Settings' section title"
37
+ OBSERVE IN THIS ORDER:
38
+ 1. Target first: identify the smallest real UI part at the callout endpoint/center: text, glyph, icon, arrow, input, dropdown/select, option, button, link, status, checkbox, radio, switch, tab, menu item, slider, image, control, or empty region.
39
+ 2. Primitive: name what that smallest part is before adding surrounding context.
40
+ 3. Owner/context: add the nearest stable owner only when it helps disambiguate, such as a label, row/card title, column header, field name, or adjacent visible text.
41
+ 4. Similar candidates: if multiple candidates look similar, add stable local anchors from the same row, card, field, header, or group. Prefer visible text and values over inferred row counting or temporary visual state.
48
42
 
49
- GUIDELINES:
50
- - Keep description under 25 words
51
- - Prioritize semantic identifiers over visual ones
52
- - Use consistent terminology across similar elements
53
- - Avoid page-specific or temporary content
54
- - Don't mention the red rectangle or selection box
55
- - Focus on stable, reusable characteristics
56
- - If the selected point/box is inside a text input, textarea, search box, or form field, describe the whole field/control, not the individual placeholder character, typed character, caret, or inner text fragment.
57
- - For icon-only buttons or unlabeled controls, include the nearest stable label, section title, menu item text, row text, or parent region that owns the control.
58
- - When multiple similar icons or controls appear in a list/sidebar/menu, the description MUST distinguish the selected one by its owning stable text or section, not by generic position such as "bottom", "nearby", or "sidebar button".
59
- - For expand/collapse, disclosure, chevron, close, menu, and settings icons, describe both the icon purpose and the stable text/section it controls.
60
- - Use the actual visible neighboring text from the current screenshot when available; do not copy labels from the examples.
43
+ RULES:
44
+ - Keep description under 35 words.
45
+ - Describe the smallest indicated UI part itself, not the larger container, row, card, sentence, or group that merely contains it.
46
+ - Ignore every annotation overlay, including callout number, line, color, marker, border, dot, ring, crosshair, or selection box. Never describe the annotation as the target.
47
+ - Do not borrow the text, glyph, direction, purpose, or state from a nearby element outside the callout endpoint/center.
48
+ - For tiny or icon-only controls, name the visible glyph/control and add its owner/context; adjacent text is context, not the target. If similar tiny controls are adjacent in the same group, add local order or relative position inside that group.
49
+ - If the endpoint/center is on a field value, label, or input body, describe that value/field/control. Do not retarget to a trailing icon, dropdown arrow, clear button, or search affordance unless the endpoint/center is on that icon itself.
50
+ - If the endpoint/center is inside the bordered body, current value, trigger, or blank area of a select/dropdown/combobox/filter field, use primitive "dropdown" and describe that dropdown/select control. Do not call it an input unless it is clearly a free-text field.
51
+ - If the endpoint/center is inside the bordered body or blank area of an input or filter field, describe the field body/current value/control even when the visible text is not exactly under the endpoint. Use the field label or visible value as context; do not snap to trailing icons or nearby table headers.
52
+ - If the endpoint/center is on an expanded dropdown/select/menu list item, use primitive "option" for selectable list options or "menuitem" for command menu entries.
53
+ - Only use primitive "icon" or "arrow" when the endpoint/center directly overlaps the real glyph strokes. A nearby search icon, dropdown arrow, clear button, or wrong locator result must not become the target primitive.
54
+ - For compound controls or stacked glyphs, describe only the sub-part containing the callout endpoint/center, using upper/lower or left/right only when visible.
55
+ - For inline text, links, or substrings, describe only the exact substring/link at the endpoint/center, not the whole line.
56
+ - For repeated rows, cards, or options, use same-local anchors that are visible in the screenshot, such as neighboring cell text, field value, title, date, time, ID, or column/header label.
57
+ - For tables or grids, describe the target as the intersection of the target column/header and same-row anchors. Do not use row ordinals or column ordinals unless the index/header is clearly visible.
58
+ - Use selected, highlighted, hovered, focused, or active state only if the callout endpoint/center is inside that state.
59
+ - If the endpoint/center is on blank space, describe the empty region/gap between stable surrounding anchors. Do not invent a nearby control.
60
+ - Use actual visible text from the current screenshot when available; do not copy labels from the examples.
61
61
  - **Write the description in ${preferredLanguage}**
62
62
 
63
63
  EXAMPLES:
64
64
  ${getExamples(preferredLanguage)}
65
65
 
66
66
  Return JSON:
67
- {
68
- "description": "unique element identifier",
69
- "error"?: "error message if any"
70
- }`;
67
+ ${describeElementReturnJsonSchema()}`;
71
68
  };
72
69
  export { elementDescriberInstruction };
73
70
 
@@ -1 +1 @@
1
- {"version":3,"file":"ai-model/prompt/describe.mjs","sources":["../../../../src/ai-model/prompt/describe.ts"],"sourcesContent":["import { getPreferredLanguage } from '@midscene/shared/env';\n\nconst examplesMap: Record<string, string[]> = {\n Chinese: [\n '\"登录表单中的\"登录\"按钮\"',\n '\"搜索输入框,placeholder 为\"请输入关键词\"\"',\n '\"顶部导航栏中文字为\"首页\"的链接\"',\n '\"联系表单中的提交按钮\"',\n '\"aria-label 为\"打开菜单\"的菜单图标\"',\n '\"左侧导航栏中当前分组标题右侧的折叠图标\"',\n ],\n English: [\n '\"Login button with text \\'Sign In\\'\"',\n '\"Search input with placeholder \\'Enter keywords\\'\"',\n '\"Navigation link with text \\'Home\\' in header\"',\n '\"Submit button in contact form\"',\n '\"Menu icon with aria-label \\'Open menu\\'\"',\n '\"Collapse icon to the right of the current section title in the left sidebar\"',\n ],\n};\n\nconst getExamples = (language: string) => {\n const examples = examplesMap[language] || examplesMap.English;\n return examples.map((e) => `- ${e}`).join('\\n');\n};\n\nexport const elementDescriberInstruction = () => {\n const preferredLanguage = getPreferredLanguage();\n\n return `\nDescribe the element in the red rectangle for precise identification.\n\nIMPORTANT: You MUST write the description in ${preferredLanguage}.\n\nCRITICAL REQUIREMENTS:\n1. UNIQUENESS: The description must uniquely identify this element on the current page\n2. UNIVERSALITY: Use generic, reusable selectors that work across different contexts\n3. PRECISION: Be specific enough to distinguish from similar elements\n\nDESCRIPTION STRUCTURE:\n1. Element type (button, input, link, div, etc.)\n2. Primary identifier (in order of preference):\n - Unique text content: \"with text 'Login'\"\n - Unique attribute: \"with aria-label 'Search'\"\n - Unique class/ID: \"with class 'primary-button'\"\n - Unique position: \"in header navigation\"\n3. Secondary identifiers (if needed for uniqueness):\n - Visual features: \"blue background\", \"with icon\"\n - Relative position: \"below search bar\", \"in sidebar\"\n - Parent context: \"in login form\", \"in main menu\"\n - Neighboring stable text: \"to the right of the 'Settings' section title\"\n\nGUIDELINES:\n- Keep description under 25 words\n- Prioritize semantic identifiers over visual ones\n- Use consistent terminology across similar elements\n- Avoid page-specific or temporary content\n- Don't mention the red rectangle or selection box\n- Focus on stable, reusable characteristics\n- If the selected point/box is inside a text input, textarea, search box, or form field, describe the whole field/control, not the individual placeholder character, typed character, caret, or inner text fragment.\n- For icon-only buttons or unlabeled controls, include the nearest stable label, section title, menu item text, row text, or parent region that owns the control.\n- When multiple similar icons or controls appear in a list/sidebar/menu, the description MUST distinguish the selected one by its owning stable text or section, not by generic position such as \"bottom\", \"nearby\", or \"sidebar button\".\n- For expand/collapse, disclosure, chevron, close, menu, and settings icons, describe both the icon purpose and the stable text/section it controls.\n- Use the actual visible neighboring text from the current screenshot when available; do not copy labels from the examples.\n- **Write the description in ${preferredLanguage}**\n\nEXAMPLES:\n${getExamples(preferredLanguage)}\n\nReturn JSON:\n{\n \"description\": \"unique element identifier\",\n \"error\"?: \"error message if any\"\n}`;\n};\n"],"names":["examplesMap","getExamples","language","examples","e","elementDescriberInstruction","preferredLanguage","getPreferredLanguage"],"mappings":";AAEA,MAAMA,cAAwC;IAC5C,SAAS;QACP;QACA;QACA;QACA;QACA;QACA;KACD;IACD,SAAS;QACP;QACA;QACA;QACA;QACA;QACA;KACD;AACH;AAEA,MAAMC,cAAc,CAACC;IACnB,MAAMC,WAAWH,WAAW,CAACE,SAAS,IAAIF,YAAY,OAAO;IAC7D,OAAOG,SAAS,GAAG,CAAC,CAACC,IAAM,CAAC,EAAE,EAAEA,GAAG,EAAE,IAAI,CAAC;AAC5C;AAEO,MAAMC,8BAA8B;IACzC,MAAMC,oBAAoBC;IAE1B,OAAO,CAAC;;;6CAGmC,EAAED,kBAAkB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;6BAgCpC,EAAEA,kBAAkB;;;AAGjD,EAAEL,YAAYK,mBAAmB;;;;;;CAMhC,CAAC;AACF"}
1
+ {"version":3,"file":"ai-model/prompt/describe.mjs","sources":["../../../../src/ai-model/prompt/describe.ts"],"sourcesContent":["import { getPreferredLanguage } from '@midscene/shared/env';\n\nconst examplesMap: Record<string, string[]> = {\n Chinese: [\n '\"登录表单中的\"登录\"按钮\"',\n '\"搜索输入框,placeholder 为\"请输入关键词\"\"',\n '\"顶部导航栏中文字为\"首页\"的链接\"',\n '\"联系表单中的提交按钮\"',\n '\"aria-label 为\"打开菜单\"的菜单图标\"',\n '\"左侧导航栏中当前分组标题右侧的折叠图标\"',\n ],\n English: [\n '\"Login button with text \\'Sign In\\'\"',\n '\"Search input with placeholder \\'Enter keywords\\'\"',\n '\"Navigation link with text \\'Home\\' in header\"',\n '\"Submit button in contact form\"',\n '\"Menu icon with aria-label \\'Open menu\\'\"',\n '\"Collapse icon to the right of the current section title in the left sidebar\"',\n ],\n};\n\nconst getExamples = (language: string) => {\n const examples = examplesMap[language] || examplesMap.English;\n return examples.map((e) => `- ${e}`).join('\\n');\n};\n\nconst describeElementReturnJsonSchema = () => `{\n \"description\": \"unique element identifier\",\n \"error\"?: \"error message if any\"\n}`;\n\nexport const elementDescriberInstruction = () => {\n const preferredLanguage = getPreferredLanguage();\n\n return `\nDescribe the real page element indicated by the temporary callout.\nThe callout is an annotation overlay. It is not part of the page or target.\nThe description will be used later to locate the same element on the original screenshot without annotations, so write a locator-style description instead of a visual caption.\n\nIMPORTANT: You MUST write the description in ${preferredLanguage}.\n\nOBSERVE IN THIS ORDER:\n1. Target first: identify the smallest real UI part at the callout endpoint/center: text, glyph, icon, arrow, input, dropdown/select, option, button, link, status, checkbox, radio, switch, tab, menu item, slider, image, control, or empty region.\n2. Primitive: name what that smallest part is before adding surrounding context.\n3. Owner/context: add the nearest stable owner only when it helps disambiguate, such as a label, row/card title, column header, field name, or adjacent visible text.\n4. Similar candidates: if multiple candidates look similar, add stable local anchors from the same row, card, field, header, or group. Prefer visible text and values over inferred row counting or temporary visual state.\n\nRULES:\n- Keep description under 35 words.\n- Describe the smallest indicated UI part itself, not the larger container, row, card, sentence, or group that merely contains it.\n- Ignore every annotation overlay, including callout number, line, color, marker, border, dot, ring, crosshair, or selection box. Never describe the annotation as the target.\n- Do not borrow the text, glyph, direction, purpose, or state from a nearby element outside the callout endpoint/center.\n- For tiny or icon-only controls, name the visible glyph/control and add its owner/context; adjacent text is context, not the target. If similar tiny controls are adjacent in the same group, add local order or relative position inside that group.\n- If the endpoint/center is on a field value, label, or input body, describe that value/field/control. Do not retarget to a trailing icon, dropdown arrow, clear button, or search affordance unless the endpoint/center is on that icon itself.\n- If the endpoint/center is inside the bordered body, current value, trigger, or blank area of a select/dropdown/combobox/filter field, use primitive \"dropdown\" and describe that dropdown/select control. Do not call it an input unless it is clearly a free-text field.\n- If the endpoint/center is inside the bordered body or blank area of an input or filter field, describe the field body/current value/control even when the visible text is not exactly under the endpoint. Use the field label or visible value as context; do not snap to trailing icons or nearby table headers.\n- If the endpoint/center is on an expanded dropdown/select/menu list item, use primitive \"option\" for selectable list options or \"menuitem\" for command menu entries.\n- Only use primitive \"icon\" or \"arrow\" when the endpoint/center directly overlaps the real glyph strokes. A nearby search icon, dropdown arrow, clear button, or wrong locator result must not become the target primitive.\n- For compound controls or stacked glyphs, describe only the sub-part containing the callout endpoint/center, using upper/lower or left/right only when visible.\n- For inline text, links, or substrings, describe only the exact substring/link at the endpoint/center, not the whole line.\n- For repeated rows, cards, or options, use same-local anchors that are visible in the screenshot, such as neighboring cell text, field value, title, date, time, ID, or column/header label.\n- For tables or grids, describe the target as the intersection of the target column/header and same-row anchors. Do not use row ordinals or column ordinals unless the index/header is clearly visible.\n- Use selected, highlighted, hovered, focused, or active state only if the callout endpoint/center is inside that state.\n- If the endpoint/center is on blank space, describe the empty region/gap between stable surrounding anchors. Do not invent a nearby control.\n- Use actual visible text from the current screenshot when available; do not copy labels from the examples.\n- **Write the description in ${preferredLanguage}**\n\nEXAMPLES:\n${getExamples(preferredLanguage)}\n\nReturn JSON:\n${describeElementReturnJsonSchema()}`;\n};\n"],"names":["examplesMap","getExamples","language","examples","e","describeElementReturnJsonSchema","elementDescriberInstruction","preferredLanguage","getPreferredLanguage"],"mappings":";AAEA,MAAMA,cAAwC;IAC5C,SAAS;QACP;QACA;QACA;QACA;QACA;QACA;KACD;IACD,SAAS;QACP;QACA;QACA;QACA;QACA;QACA;KACD;AACH;AAEA,MAAMC,cAAc,CAACC;IACnB,MAAMC,WAAWH,WAAW,CAACE,SAAS,IAAIF,YAAY,OAAO;IAC7D,OAAOG,SAAS,GAAG,CAAC,CAACC,IAAM,CAAC,EAAE,EAAEA,GAAG,EAAE,IAAI,CAAC;AAC5C;AAEA,MAAMC,kCAAkC,IAAM,CAAC;;;CAG9C,CAAC;AAEK,MAAMC,8BAA8B;IACzC,MAAMC,oBAAoBC;IAE1B,OAAO,CAAC;;;;;6CAKmC,EAAED,kBAAkB;;;;;;;;;;;;;;;;;;;;;;;;;;6BA0BpC,EAAEA,kBAAkB;;;AAGjD,EAAEN,YAAYM,mBAAmB;;;AAGjC,EAAEF,mCAAmC;AACrC"}
@@ -1,5 +1,5 @@
1
1
  import { getPreferredLanguage } from "@midscene/shared/env";
2
- import { safeParseJson } from "../service-caller/json.mjs";
2
+ import { parseModelResponseJson } from "../service-caller/json.mjs";
3
3
  import { extractXMLTag } from "./util.mjs";
4
4
  function buildTypeQueryDemandValue(type, demand) {
5
5
  const currentScreenshotConstraint = 'based on the current screenshot and its contents if provided, unless the user explicitly asks to compare with reference images';
@@ -8,19 +8,25 @@ function buildTypeQueryDemandValue(type, demand) {
8
8
  return `${type}, ${currentScreenshotConstraint}, ${demand}`;
9
9
  }
10
10
  function parseXMLExtractionResponse(xmlString) {
11
- const thought = extractXMLTag(xmlString, 'thought');
11
+ const thought = extractXMLTag(xmlString, 'observation');
12
12
  const dataJsonStr = extractXMLTag(xmlString, 'data-json');
13
13
  const errorsStr = extractXMLTag(xmlString, 'errors');
14
14
  if (!dataJsonStr) throw new Error('Missing required field: data-json');
15
15
  let data;
16
16
  try {
17
- data = safeParseJson(dataJsonStr);
17
+ data = parseModelResponseJson(dataJsonStr, {
18
+ source: 'generic-object',
19
+ requireObject: false
20
+ });
18
21
  } catch (e) {
19
22
  throw new Error(`Failed to parse data-json: ${e}`);
20
23
  }
21
24
  let errors;
22
25
  if (errorsStr) try {
23
- const parsedErrors = safeParseJson(errorsStr);
26
+ const parsedErrors = parseModelResponseJson(errorsStr, {
27
+ source: 'generic-object',
28
+ requireObject: false
29
+ });
24
30
  if (Array.isArray(parsedErrors)) errors = parsedErrors;
25
31
  } catch (e) {}
26
32
  return {
@@ -58,7 +64,7 @@ When DATA_DEMAND is a JSON object, the keys in your response must exactly match
58
64
 
59
65
 
60
66
  Return in the following XML format:
61
- <thought>the thinking process of the extraction, less than 300 words. Use ${preferredLanguage} in this field.</thought>
67
+ <observation>brief evidence observed for the extraction, less than 300 words. Use ${preferredLanguage} in this field.</observation>
62
68
  <data-json>the extracted data as JSON. Make sure both the value and scheme meet the DATA_DEMAND. If you want to write some description in this field, use the same language as the DATA_DEMAND.</data-json>
63
69
  <errors>optional error messages as JSON array, e.g., ["error1", "error2"]</errors>
64
70
 
@@ -75,7 +81,7 @@ For example, if the DATA_DEMAND is:
75
81
 
76
82
  By viewing the screenshot and page contents, you can extract the following data:
77
83
 
78
- <thought>According to the screenshot, i can see ...</thought>
84
+ <observation>According to the screenshot, i can see ...</observation>
79
85
  <data-json>
80
86
  {
81
87
  "name": "John",
@@ -93,7 +99,7 @@ the todo items list, string[]
93
99
 
94
100
  By viewing the screenshot and page contents, you can extract the following data:
95
101
 
96
- <thought>According to the screenshot, i can see ...</thought>
102
+ <observation>According to the screenshot, i can see ...</observation>
97
103
  <data-json>
98
104
  ["todo 1", "todo 2", "todo 3"]
99
105
  </data-json>
@@ -107,7 +113,7 @@ the page title, string
107
113
 
108
114
  By viewing the screenshot and page contents, you can extract the following data:
109
115
 
110
- <thought>According to the screenshot, i can see ...</thought>
116
+ <observation>According to the screenshot, i can see ...</observation>
111
117
  <data-json>
112
118
  "todo list"
113
119
  </data-json>
@@ -123,7 +129,7 @@ If the DATA_DEMAND is:
123
129
 
124
130
  By viewing the screenshot and page contents, you can extract the following data:
125
131
 
126
- <thought>According to the screenshot, i can see ...</thought>
132
+ <observation>According to the screenshot, i can see ...</observation>
127
133
  <data-json>
128
134
  { "StatementIsTruthy": true }
129
135
  </data-json>
@@ -1 +1 @@
1
- {"version":3,"file":"ai-model/prompt/extraction.mjs","sources":["../../../../src/ai-model/prompt/extraction.ts"],"sourcesContent":["import type { AIDataExtractionResponse, ServiceExtractParam } from '@/types';\nimport { getPreferredLanguage } from '@midscene/shared/env';\nimport { safeParseJson } from '../service-caller/json';\nimport { extractXMLTag } from './util';\n\nexport function buildTypeQueryDemandValue(\n type: 'Boolean' | 'Number' | 'String' | 'Assert' | 'WaitFor',\n demand: ServiceExtractParam,\n) {\n const currentScreenshotConstraint =\n 'based on the current screenshot and its contents if provided, unless the user explicitly asks to compare with reference images';\n\n if (type === 'Assert') {\n return `Boolean, ${currentScreenshotConstraint}, whether the following statement is true: ${demand}`;\n }\n\n if (type === 'WaitFor') {\n return `Boolean, the user wants to do some 'wait for' operation. ${currentScreenshotConstraint}, please check whether the following statement is true: ${demand}`;\n }\n\n return `${type}, ${currentScreenshotConstraint}, ${demand}`;\n}\n\n/**\n * Parse XML response from LLM and convert to AIDataExtractionResponse\n */\nexport function parseXMLExtractionResponse<T>(\n xmlString: string,\n): AIDataExtractionResponse<T> {\n const thought = extractXMLTag(xmlString, 'thought');\n const dataJsonStr = extractXMLTag(xmlString, 'data-json');\n const errorsStr = extractXMLTag(xmlString, 'errors');\n\n // Parse data-json (required)\n if (!dataJsonStr) {\n throw new Error('Missing required field: data-json');\n }\n\n let data: T;\n try {\n data = safeParseJson(dataJsonStr) as T;\n } catch (e) {\n throw new Error(`Failed to parse data-json: ${e}`);\n }\n\n // Parse errors (optional)\n let errors: string[] | undefined;\n if (errorsStr) {\n try {\n const parsedErrors = safeParseJson(errorsStr);\n if (Array.isArray(parsedErrors)) {\n errors = parsedErrors;\n }\n } catch (e) {\n // If errors parsing fails, just ignore it\n }\n }\n\n return {\n ...(thought ? { thought } : {}),\n data,\n ...(errors && errors.length > 0 ? { errors } : {}),\n };\n}\n\nexport function systemPromptToExtract(options?: {\n screenshotIncluded?: boolean;\n referenceImagesIncluded?: boolean;\n}) {\n const preferredLanguage = getPreferredLanguage();\n const screenshotIncluded = options?.screenshotIncluded ?? true;\n const referenceImagesIncluded = options?.referenceImagesIncluded ?? false;\n\n const contextPrompts = [\n \"The user will give you data requirements in <DATA_DEMAND>. You need to understand the user's requirements and extract the data satisfying the <DATA_DEMAND>.\",\n ];\n\n if (screenshotIncluded) {\n contextPrompts.push(\n 'The user will provide a current screenshot to evaluate, and may provide its contents. Base your answer on the current screenshot and its contents when provided. Treat them as the primary source of truth for what is currently visible or true.',\n );\n } else {\n contextPrompts.push(\n 'The user will not provide a current screenshot. Use only the supplied page contents and other inputs, and do not infer unsupported visual details.',\n );\n }\n\n if (referenceImagesIncluded) {\n const referenceImagesPrompt =\n 'Reference images are supporting context only unless <DATA_DEMAND> explicitly asks for comparison, matching, or reasoning about them.';\n contextPrompts.push(\n screenshotIncluded\n ? `${referenceImagesPrompt} Do not conclude that something exists in the current screenshot solely because it appears in a reference image; when they conflict, trust the current screenshot and its contents.`\n : `${referenceImagesPrompt} Do not treat reference images as direct evidence of the current state unless the demand explicitly asks you to use them that way.`,\n );\n }\n const contextPrompt = contextPrompts.join('\\n\\n');\n\n return `\nYou are a versatile professional in software UI design and testing. Your outstanding contributions will impact the user experience of billions of users.\n\n${contextPrompt}\n\nIf a key specifies a JSON data type (such as Number, String, Boolean, Object, Array), ensure the returned value strictly matches that data type.\n\nWhen DATA_DEMAND is a JSON object, the keys in your response must exactly match the keys in DATA_DEMAND. Do not rename, translate, or substitute any key.\n\n\nReturn in the following XML format:\n<thought>the thinking process of the extraction, less than 300 words. Use ${preferredLanguage} in this field.</thought>\n<data-json>the extracted data as JSON. Make sure both the value and scheme meet the DATA_DEMAND. If you want to write some description in this field, use the same language as the DATA_DEMAND.</data-json>\n<errors>optional error messages as JSON array, e.g., [\"error1\", \"error2\"]</errors>\n\n# Example 1\nFor example, if the DATA_DEMAND is:\n\n<DATA_DEMAND>\n{\n \"name\": \"name shows on the left panel, string\",\n \"age\": \"age shows on the right panel, number\",\n \"isAdmin\": \"if the user is admin, boolean\"\n}\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<thought>According to the screenshot, i can see ...</thought>\n<data-json>\n{\n \"name\": \"John\",\n \"age\": 30,\n \"isAdmin\": true\n}\n</data-json>\n\n# Example 2\nIf the DATA_DEMAND is:\n\n<DATA_DEMAND>\nthe todo items list, string[]\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<thought>According to the screenshot, i can see ...</thought>\n<data-json>\n[\"todo 1\", \"todo 2\", \"todo 3\"]\n</data-json>\n\n# Example 3\nIf the DATA_DEMAND is:\n\n<DATA_DEMAND>\nthe page title, string\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<thought>According to the screenshot, i can see ...</thought>\n<data-json>\n\"todo list\"\n</data-json>\n\n# Example 4\nIf the DATA_DEMAND is:\n\n<DATA_DEMAND>\n{\n \"StatementIsTruthy\": \"Boolean, is it currently the SMS page?\"\n}\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<thought>According to the screenshot, i can see ...</thought>\n<data-json>\n{ \"StatementIsTruthy\": true }\n</data-json>\n`;\n}\n\nexport const extractDataQueryPrompt = (\n pageDescription: string,\n dataQuery: string | Record<string, string>,\n) => {\n let dataQueryText = '';\n if (typeof dataQuery === 'string') {\n dataQueryText = dataQuery;\n } else {\n dataQueryText = JSON.stringify(dataQuery, null, 2);\n }\n\n return `\n<PageDescription>\n${pageDescription}\n</PageDescription>\n\n<DATA_DEMAND>\n${dataQueryText}\n</DATA_DEMAND>\n `;\n};\n"],"names":["buildTypeQueryDemandValue","type","demand","currentScreenshotConstraint","parseXMLExtractionResponse","xmlString","thought","extractXMLTag","dataJsonStr","errorsStr","Error","data","safeParseJson","e","errors","parsedErrors","Array","systemPromptToExtract","options","preferredLanguage","getPreferredLanguage","screenshotIncluded","referenceImagesIncluded","contextPrompts","referenceImagesPrompt","contextPrompt","extractDataQueryPrompt","pageDescription","dataQuery","dataQueryText","JSON"],"mappings":";;;AAKO,SAASA,0BACdC,IAA4D,EAC5DC,MAA2B;IAE3B,MAAMC,8BACJ;IAEF,IAAIF,AAAS,aAATA,MACF,OAAO,CAAC,SAAS,EAAEE,4BAA4B,2CAA2C,EAAED,QAAQ;IAGtG,IAAID,AAAS,cAATA,MACF,OAAO,CAAC,yDAAyD,EAAEE,4BAA4B,wDAAwD,EAAED,QAAQ;IAGnK,OAAO,GAAGD,KAAK,EAAE,EAAEE,4BAA4B,EAAE,EAAED,QAAQ;AAC7D;AAKO,SAASE,2BACdC,SAAiB;IAEjB,MAAMC,UAAUC,cAAcF,WAAW;IACzC,MAAMG,cAAcD,cAAcF,WAAW;IAC7C,MAAMI,YAAYF,cAAcF,WAAW;IAG3C,IAAI,CAACG,aACH,MAAM,IAAIE,MAAM;IAGlB,IAAIC;IACJ,IAAI;QACFA,OAAOC,cAAcJ;IACvB,EAAE,OAAOK,GAAG;QACV,MAAM,IAAIH,MAAM,CAAC,2BAA2B,EAAEG,GAAG;IACnD;IAGA,IAAIC;IACJ,IAAIL,WACF,IAAI;QACF,MAAMM,eAAeH,cAAcH;QACnC,IAAIO,MAAM,OAAO,CAACD,eAChBD,SAASC;IAEb,EAAE,OAAOF,GAAG,CAEZ;IAGF,OAAO;QACL,GAAIP,UAAU;YAAEA;QAAQ,IAAI,CAAC,CAAC;QAC9BK;QACA,GAAIG,UAAUA,OAAO,MAAM,GAAG,IAAI;YAAEA;QAAO,IAAI,CAAC,CAAC;IACnD;AACF;AAEO,SAASG,sBAAsBC,OAGrC;IACC,MAAMC,oBAAoBC;IAC1B,MAAMC,qBAAqBH,SAAS,sBAAsB;IAC1D,MAAMI,0BAA0BJ,SAAS,2BAA2B;IAEpE,MAAMK,iBAAiB;QACrB;KACD;IAED,IAAIF,oBACFE,eAAe,IAAI,CACjB;SAGFA,eAAe,IAAI,CACjB;IAIJ,IAAID,yBAAyB;QAC3B,MAAME,wBACJ;QACFD,eAAe,IAAI,CACjBF,qBACI,GAAGG,sBAAsB,mLAAmL,CAAC,GAC7M,GAAGA,sBAAsB,kIAAkI,CAAC;IAEpK;IACA,MAAMC,gBAAgBF,eAAe,IAAI,CAAC;IAE1C,OAAO,CAAC;;;AAGV,EAAEE,cAAc;;;;;;;;0EAQ0D,EAAEN,kBAAkB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqE9F,CAAC;AACD;AAEO,MAAMO,yBAAyB,CACpCC,iBACAC;IAEA,IAAIC,gBAAgB;IAElBA,gBADE,AAAqB,YAArB,OAAOD,YACOA,YAEAE,KAAK,SAAS,CAACF,WAAW,MAAM;IAGlD,OAAO,CAAC;;AAEV,EAAED,gBAAgB;;;;AAIlB,EAAEE,cAAc;;EAEd,CAAC;AACH"}
1
+ {"version":3,"file":"ai-model/prompt/extraction.mjs","sources":["../../../../src/ai-model/prompt/extraction.ts"],"sourcesContent":["import type { AIDataExtractionResponse, ServiceExtractParam } from '@/types';\nimport { getPreferredLanguage } from '@midscene/shared/env';\nimport { parseModelResponseJson } from '../service-caller/json';\nimport { extractXMLTag } from './util';\n\nexport function buildTypeQueryDemandValue(\n type: 'Boolean' | 'Number' | 'String' | 'Assert' | 'WaitFor',\n demand: ServiceExtractParam,\n) {\n const currentScreenshotConstraint =\n 'based on the current screenshot and its contents if provided, unless the user explicitly asks to compare with reference images';\n\n if (type === 'Assert') {\n return `Boolean, ${currentScreenshotConstraint}, whether the following statement is true: ${demand}`;\n }\n\n if (type === 'WaitFor') {\n return `Boolean, the user wants to do some 'wait for' operation. ${currentScreenshotConstraint}, please check whether the following statement is true: ${demand}`;\n }\n\n return `${type}, ${currentScreenshotConstraint}, ${demand}`;\n}\n\n/**\n * Parse XML response from LLM and convert to AIDataExtractionResponse\n */\nexport function parseXMLExtractionResponse<T>(\n xmlString: string,\n): AIDataExtractionResponse<T> {\n // Keep the internal field named `thought`, but ask models to emit\n // <observation>. Gemini may only return <thought>-named content when\n // thinking summaries are enabled.\n const thought = extractXMLTag(xmlString, 'observation');\n const dataJsonStr = extractXMLTag(xmlString, 'data-json');\n const errorsStr = extractXMLTag(xmlString, 'errors');\n\n // Parse data-json (required)\n if (!dataJsonStr) {\n throw new Error('Missing required field: data-json');\n }\n\n let data: T;\n try {\n data = parseModelResponseJson(dataJsonStr, {\n source: 'generic-object',\n requireObject: false,\n }) as T;\n } catch (e) {\n throw new Error(`Failed to parse data-json: ${e}`);\n }\n\n // Parse errors (optional)\n let errors: string[] | undefined;\n if (errorsStr) {\n try {\n const parsedErrors = parseModelResponseJson(errorsStr, {\n source: 'generic-object',\n requireObject: false,\n });\n if (Array.isArray(parsedErrors)) {\n errors = parsedErrors;\n }\n } catch (e) {\n // If errors parsing fails, just ignore it\n }\n }\n\n return {\n ...(thought ? { thought } : {}),\n data,\n ...(errors && errors.length > 0 ? { errors } : {}),\n };\n}\n\nexport function systemPromptToExtract(options?: {\n screenshotIncluded?: boolean;\n referenceImagesIncluded?: boolean;\n}) {\n const preferredLanguage = getPreferredLanguage();\n const screenshotIncluded = options?.screenshotIncluded ?? true;\n const referenceImagesIncluded = options?.referenceImagesIncluded ?? false;\n\n const contextPrompts = [\n \"The user will give you data requirements in <DATA_DEMAND>. You need to understand the user's requirements and extract the data satisfying the <DATA_DEMAND>.\",\n ];\n\n if (screenshotIncluded) {\n contextPrompts.push(\n 'The user will provide a current screenshot to evaluate, and may provide its contents. Base your answer on the current screenshot and its contents when provided. Treat them as the primary source of truth for what is currently visible or true.',\n );\n } else {\n contextPrompts.push(\n 'The user will not provide a current screenshot. Use only the supplied page contents and other inputs, and do not infer unsupported visual details.',\n );\n }\n\n if (referenceImagesIncluded) {\n const referenceImagesPrompt =\n 'Reference images are supporting context only unless <DATA_DEMAND> explicitly asks for comparison, matching, or reasoning about them.';\n contextPrompts.push(\n screenshotIncluded\n ? `${referenceImagesPrompt} Do not conclude that something exists in the current screenshot solely because it appears in a reference image; when they conflict, trust the current screenshot and its contents.`\n : `${referenceImagesPrompt} Do not treat reference images as direct evidence of the current state unless the demand explicitly asks you to use them that way.`,\n );\n }\n const contextPrompt = contextPrompts.join('\\n\\n');\n\n return `\nYou are a versatile professional in software UI design and testing. Your outstanding contributions will impact the user experience of billions of users.\n\n${contextPrompt}\n\nIf a key specifies a JSON data type (such as Number, String, Boolean, Object, Array), ensure the returned value strictly matches that data type.\n\nWhen DATA_DEMAND is a JSON object, the keys in your response must exactly match the keys in DATA_DEMAND. Do not rename, translate, or substitute any key.\n\n\nReturn in the following XML format:\n<observation>brief evidence observed for the extraction, less than 300 words. Use ${preferredLanguage} in this field.</observation>\n<data-json>the extracted data as JSON. Make sure both the value and scheme meet the DATA_DEMAND. If you want to write some description in this field, use the same language as the DATA_DEMAND.</data-json>\n<errors>optional error messages as JSON array, e.g., [\"error1\", \"error2\"]</errors>\n\n# Example 1\nFor example, if the DATA_DEMAND is:\n\n<DATA_DEMAND>\n{\n \"name\": \"name shows on the left panel, string\",\n \"age\": \"age shows on the right panel, number\",\n \"isAdmin\": \"if the user is admin, boolean\"\n}\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<observation>According to the screenshot, i can see ...</observation>\n<data-json>\n{\n \"name\": \"John\",\n \"age\": 30,\n \"isAdmin\": true\n}\n</data-json>\n\n# Example 2\nIf the DATA_DEMAND is:\n\n<DATA_DEMAND>\nthe todo items list, string[]\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<observation>According to the screenshot, i can see ...</observation>\n<data-json>\n[\"todo 1\", \"todo 2\", \"todo 3\"]\n</data-json>\n\n# Example 3\nIf the DATA_DEMAND is:\n\n<DATA_DEMAND>\nthe page title, string\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<observation>According to the screenshot, i can see ...</observation>\n<data-json>\n\"todo list\"\n</data-json>\n\n# Example 4\nIf the DATA_DEMAND is:\n\n<DATA_DEMAND>\n{\n \"StatementIsTruthy\": \"Boolean, is it currently the SMS page?\"\n}\n</DATA_DEMAND>\n\nBy viewing the screenshot and page contents, you can extract the following data:\n\n<observation>According to the screenshot, i can see ...</observation>\n<data-json>\n{ \"StatementIsTruthy\": true }\n</data-json>\n`;\n}\n\nexport const extractDataQueryPrompt = (\n pageDescription: string,\n dataQuery: string | Record<string, string>,\n) => {\n let dataQueryText = '';\n if (typeof dataQuery === 'string') {\n dataQueryText = dataQuery;\n } else {\n dataQueryText = JSON.stringify(dataQuery, null, 2);\n }\n\n return `\n<PageDescription>\n${pageDescription}\n</PageDescription>\n\n<DATA_DEMAND>\n${dataQueryText}\n</DATA_DEMAND>\n `;\n};\n"],"names":["buildTypeQueryDemandValue","type","demand","currentScreenshotConstraint","parseXMLExtractionResponse","xmlString","thought","extractXMLTag","dataJsonStr","errorsStr","Error","data","parseModelResponseJson","e","errors","parsedErrors","Array","systemPromptToExtract","options","preferredLanguage","getPreferredLanguage","screenshotIncluded","referenceImagesIncluded","contextPrompts","referenceImagesPrompt","contextPrompt","extractDataQueryPrompt","pageDescription","dataQuery","dataQueryText","JSON"],"mappings":";;;AAKO,SAASA,0BACdC,IAA4D,EAC5DC,MAA2B;IAE3B,MAAMC,8BACJ;IAEF,IAAIF,AAAS,aAATA,MACF,OAAO,CAAC,SAAS,EAAEE,4BAA4B,2CAA2C,EAAED,QAAQ;IAGtG,IAAID,AAAS,cAATA,MACF,OAAO,CAAC,yDAAyD,EAAEE,4BAA4B,wDAAwD,EAAED,QAAQ;IAGnK,OAAO,GAAGD,KAAK,EAAE,EAAEE,4BAA4B,EAAE,EAAED,QAAQ;AAC7D;AAKO,SAASE,2BACdC,SAAiB;IAKjB,MAAMC,UAAUC,cAAcF,WAAW;IACzC,MAAMG,cAAcD,cAAcF,WAAW;IAC7C,MAAMI,YAAYF,cAAcF,WAAW;IAG3C,IAAI,CAACG,aACH,MAAM,IAAIE,MAAM;IAGlB,IAAIC;IACJ,IAAI;QACFA,OAAOC,uBAAuBJ,aAAa;YACzC,QAAQ;YACR,eAAe;QACjB;IACF,EAAE,OAAOK,GAAG;QACV,MAAM,IAAIH,MAAM,CAAC,2BAA2B,EAAEG,GAAG;IACnD;IAGA,IAAIC;IACJ,IAAIL,WACF,IAAI;QACF,MAAMM,eAAeH,uBAAuBH,WAAW;YACrD,QAAQ;YACR,eAAe;QACjB;QACA,IAAIO,MAAM,OAAO,CAACD,eAChBD,SAASC;IAEb,EAAE,OAAOF,GAAG,CAEZ;IAGF,OAAO;QACL,GAAIP,UAAU;YAAEA;QAAQ,IAAI,CAAC,CAAC;QAC9BK;QACA,GAAIG,UAAUA,OAAO,MAAM,GAAG,IAAI;YAAEA;QAAO,IAAI,CAAC,CAAC;IACnD;AACF;AAEO,SAASG,sBAAsBC,OAGrC;IACC,MAAMC,oBAAoBC;IAC1B,MAAMC,qBAAqBH,SAAS,sBAAsB;IAC1D,MAAMI,0BAA0BJ,SAAS,2BAA2B;IAEpE,MAAMK,iBAAiB;QACrB;KACD;IAED,IAAIF,oBACFE,eAAe,IAAI,CACjB;SAGFA,eAAe,IAAI,CACjB;IAIJ,IAAID,yBAAyB;QAC3B,MAAME,wBACJ;QACFD,eAAe,IAAI,CACjBF,qBACI,GAAGG,sBAAsB,mLAAmL,CAAC,GAC7M,GAAGA,sBAAsB,kIAAkI,CAAC;IAEpK;IACA,MAAMC,gBAAgBF,eAAe,IAAI,CAAC;IAE1C,OAAO,CAAC;;;AAGV,EAAEE,cAAc;;;;;;;;kFAQkE,EAAEN,kBAAkB;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAqEtG,CAAC;AACD;AAEO,MAAMO,yBAAyB,CACpCC,iBACAC;IAEA,IAAIC,gBAAgB;IAElBA,gBADE,AAAqB,YAArB,OAAOD,YACOA,YAEAE,KAAK,SAAS,CAACF,WAAW,MAAM;IAGlD,OAAO,CAAC;;AAEV,EAAED,gBAAgB;;;;AAIlB,EAAEE,cAAc;;EAEd,CAAC;AACH"}
@@ -112,16 +112,16 @@ async function systemPromptToTaskPlanning({ actionSpace, locatePromptSpec, inclu
112
112
  const locateExample1 = locateExample('Add to cart button for Sauce Labs Backpack', 1);
113
113
  const locateNameField = locateExample('Name input field in the registration form', 2);
114
114
  const locateEmailField = locateExample('Email input field in the registration form', 3);
115
- const step1Title = shouldIncludeSubGoals ? '## Step 1: Observe and Plan (related tags: <thought>, <update-plan-content>, <mark-sub-goal-done>)' : '## Step 1: Observe (related tags: <thought>)';
115
+ const step1Title = shouldIncludeSubGoals ? '## Step 1: Observe and Plan (related tags: <planning>, <update-plan-content>, <mark-sub-goal-done>)' : '## Step 1: Observe (related tags: <planning>)';
116
116
  const step1Description = shouldIncludeSubGoals ? "First, observe the current screenshot and previous logs, then break down the user's instruction into multiple high-level sub-goals. Update the status of sub-goals based on what you see in the current screenshot." : 'First, observe the current screenshot and previous logs to understand the current state.';
117
117
  const explicitInstructionRule = 'CRITICAL - Following Explicit Instructions: When the user gives you specific operation steps (not high-level goals), you MUST execute ONLY those exact steps - nothing more, nothing less. Do NOT add extra actions even if they seem logical. For example: "fill out the form" means only fill fields, do NOT submit; "click the button" means only click, do NOT wait for page load or verify results; "type \'hello\'" means only type, do NOT press Enter.';
118
- const thoughtTagDescription = shouldIncludeSubGoals ? `REQUIRED: You MUST always output the <thought> tag. Never skip it.
118
+ const planningTagDescription = shouldIncludeSubGoals ? `REQUIRED: You MUST always output the <planning> tag. Never skip it.
119
119
 
120
- Include your thought process in the <thought> tag. It should answer: What is the user's requirement? What is the current state based on the screenshot? Are all sub-goals completed? If not, what should be the next action? Write your thoughts naturally without numbering or section headers.
120
+ Include your planning details in the <planning> tag. It should answer: What is the user's requirement? What is the current state based on the screenshot? Are all sub-goals completed? If not, what should be the next action? Write it naturally without numbering or section headers.
121
121
 
122
- ${explicitInstructionRule}` : `REQUIRED: You MUST always output the <thought> tag. Never skip it.
122
+ ${explicitInstructionRule}` : `REQUIRED: You MUST always output the <planning> tag. Never skip it.
123
123
 
124
- Include your thought process in the <thought> tag. It should answer: What is the current state based on the screenshot? What should be the next action? Write your thoughts naturally without numbering or section headers.
124
+ Include your planning details in the <planning> tag. It should answer: What is the current state based on the screenshot? What should be the next action? Write it naturally without numbering or section headers.
125
125
 
126
126
  ${explicitInstructionRule}`;
127
127
  const subGoalTags = shouldIncludeSubGoals ? `
@@ -154,7 +154,7 @@ During execution, you can call <update-plan-content> at any time to update the p
154
154
 
155
155
  If the user wants to "log in to a system using username and password, complete all to-do items, and submit a registration form", you can break it down into the following sub-goals:
156
156
 
157
- <thought>...</thought>
157
+ <planning>...</planning>
158
158
  <update-plan-content>
159
159
  <sub-goal index="1" status="pending">Log in to the system</sub-goal>
160
160
  <sub-goal index="2" status="pending">Complete all to-do items</sub-goal>
@@ -190,9 +190,9 @@ ${step1Title}
190
190
 
191
191
  ${step1Description}
192
192
  ${shouldIncludeSubGoals ? `\n${OBSERVE_STEP_NOTES}\n` : ''}
193
- * <thought> tag (REQUIRED)
193
+ * <planning> tag (REQUIRED)
194
194
 
195
- ${thoughtTagDescription}
195
+ ${planningTagDescription}
196
196
  ${subGoalTags}
197
197
  ${shouldIncludeSubGoals ? `
198
198
  ## Step ${memoryStepNumber}: Memory Data from Current Screenshot (related tags: <memory>)
@@ -335,7 +335,7 @@ Return in XML format following this decision flow:
335
335
 
336
336
  **Always include (REQUIRED):**
337
337
  <!-- Step 1: Observe${shouldIncludeSubGoals ? ' and Plan' : ''} -->
338
- <thought>Your thought process here. NEVER skip this tag.</thought>
338
+ <planning>Your planning details here. NEVER skip this tag.</planning>
339
339
  ${shouldIncludeSubGoals ? `
340
340
  <!-- required when no update-plan-content is provided in the previous response -->
341
341
  <update-plan-content>...</update-plan-content>
@@ -374,7 +374,7 @@ Below is an example of a multi-turn conversation for "fill out the registration
374
374
  **Screenshot:** [Shows a registration form with empty Name and Email fields]
375
375
 
376
376
  **Your response:**
377
- <thought>The user wants me to fill out the registration form with specific values and return the email address. I can see the form has two fields: Name and Email. Both are currently empty. I'll break this down into sub-goals and start with the Name field. Note: The instruction is to fill the form only (not submit), and return the email at the end.</thought>
377
+ <planning>The user wants me to fill out the registration form with specific values and return the email address. I can see the form has two fields: Name and Email. Both are currently empty. I'll break this down into sub-goals and start with the Name field. Note: The instruction is to fill the form only (not submit), and return the email at the end.</planning>
378
378
  <update-plan-content>
379
379
  <sub-goal index="1" status="pending">Fill in the Name field with 'John'</sub-goal>
380
380
  <sub-goal index="2" status="pending">Fill in the Email field with 'john@example.com'</sub-goal>
@@ -404,7 +404,7 @@ Actions performed for current sub-goal:
404
404
  **Screenshot:** [Shows the form with Name field now focused/active]
405
405
 
406
406
  **Your response:**
407
- <thought>The Name field is now focused. I need to type 'John' into this field. Current sub-goal is running, will be completed after input.</thought>
407
+ <planning>The Name field is now focused. I need to type 'John' into this field. Current sub-goal is running, will be completed after input.</planning>
408
408
  <log>Typing 'John' into the Name field</log>
409
409
  <action-type>Input</action-type>
410
410
  <action-param-json>
@@ -430,7 +430,7 @@ Actions performed for current sub-goal:
430
430
  **Screenshot:** [Shows the form with Name field containing 'John']
431
431
 
432
432
  **Your response:**
433
- <thought>The Name field now shows 'John'. Sub-goal 1 is complete. Next, I need to fill the Email field for sub-goal 2.</thought>
433
+ <planning>The Name field now shows 'John'. Sub-goal 1 is complete. Next, I need to fill the Email field for sub-goal 2.</planning>
434
434
  <mark-sub-goal-done>
435
435
  <sub-goal index="1" status="finished" />
436
436
  </mark-sub-goal-done>
@@ -459,7 +459,7 @@ Actions performed for current sub-goal:
459
459
  **Screenshot:** [Shows the form with Name='John' and Email field focused]
460
460
 
461
461
  **Your response:**
462
- <thought>The Email field is now focused. I'll enter 'john@example.com'. After this, sub-goal 2 will be complete and my task will be done.</thought>
462
+ <planning>The Email field is now focused. I'll enter 'john@example.com'. After this, sub-goal 2 will be complete and my task will be done.</planning>
463
463
  <log>Typing email address into the Email field</log>
464
464
  <action-type>Input</action-type>
465
465
  <action-param-json>
@@ -485,7 +485,7 @@ Actions performed for current sub-goal:
485
485
  **Screenshot:** [Shows the form with Name='John' and Email='john@example.com']
486
486
 
487
487
  **Your response:**
488
- <thought>Both fields are now filled: Name shows 'John' and Email shows 'john@example.com'. Sub-goal 2 is complete. The user asked me to return the filled email address, so I need to include 'john@example.com' in my response. All sub-goals are now finished.</thought>
488
+ <planning>Both fields are now filled: Name shows 'John' and Email shows 'john@example.com'. Sub-goal 2 is complete. The user asked me to return the filled email address, so I need to include 'john@example.com' in my response. All sub-goals are now finished.</planning>
489
489
  <mark-sub-goal-done>
490
490
  <sub-goal index="2" status="finished" />
491
491
  <sub-goal index="3" status="finished" />
@@ -504,7 +504,7 @@ Below is an example of a multi-turn conversation for "fill out the registration
504
504
  **Screenshot:** [Shows a registration form with empty Name and Email fields]
505
505
 
506
506
  **Your response:**
507
- <thought>The user wants me to fill out the registration form with specific values and return the email address. I can see the form has two fields: Name and Email. Both are currently empty. I should start by clicking on the Name field. Note: The instruction is to fill the form only (not submit), and return the email at the end.</thought>
507
+ <planning>The user wants me to fill out the registration form with specific values and return the email address. I can see the form has two fields: Name and Email. Both are currently empty. I should start by clicking on the Name field. Note: The instruction is to fill the form only (not submit), and return the email at the end.</planning>
508
508
  <log>Click on the Name field to start filling the form</log>
509
509
  <action-type>Tap</action-type>
510
510
  <action-param-json>
@@ -521,7 +521,7 @@ The previous action has been executed, here is the latest screenshot. Please con
521
521
  **Screenshot:** [Shows the form with Name field now focused/active]
522
522
 
523
523
  **Your response:**
524
- <thought>The Name field is now focused. I need to type 'John' into this field. After this, I'll need to fill the Email field as well.</thought>
524
+ <planning>The Name field is now focused. I need to type 'John' into this field. After this, I'll need to fill the Email field as well.</planning>
525
525
  <log>Typing 'John' into the Name field</log>
526
526
  <action-type>Input</action-type>
527
527
  <action-param-json>
@@ -538,7 +538,7 @@ The previous action has been executed, here is the latest screenshot. Please con
538
538
  **Screenshot:** [Shows the form with Name field containing 'John']
539
539
 
540
540
  **Your response:**
541
- <thought>The Name field now shows 'John'. Next, I need to fill the Email field with 'john@example.com'. I'll click on the Email field first.</thought>
541
+ <planning>The Name field now shows 'John'. Next, I need to fill the Email field with 'john@example.com'. I'll click on the Email field first.</planning>
542
542
  <log>Moving to the Email field</log>
543
543
  <action-type>Tap</action-type>
544
544
  <action-param-json>
@@ -555,7 +555,7 @@ The previous action has been executed, here is the latest screenshot. Please con
555
555
  **Screenshot:** [Shows the form with Name='John' and Email field focused]
556
556
 
557
557
  **Your response:**
558
- <thought>The Email field is now focused. I'll enter 'john@example.com'. After this, both required fields will be filled and my task will be complete.</thought>
558
+ <planning>The Email field is now focused. I'll enter 'john@example.com'. After this, both required fields will be filled and my task will be complete.</planning>
559
559
  <log>Typing email address into the Email field</log>
560
560
  <action-type>Input</action-type>
561
561
  <action-param-json>
@@ -572,7 +572,7 @@ The previous action has been executed, here is the latest screenshot. Please con
572
572
  **Screenshot:** [Shows the form with Name='John' and Email='john@example.com']
573
573
 
574
574
  **Your response:**
575
- <thought>Both fields are now filled: Name shows 'John' and Email shows 'john@example.com'. The user asked me to return the filled email address, so I should include 'john@example.com' in my response. The instruction has been fulfilled.</thought>
575
+ <planning>Both fields are now filled: Name shows 'John' and Email shows 'john@example.com'. The user asked me to return the filled email address, so I should include 'john@example.com' in my response. The instruction has been fulfilled.</planning>
576
576
  <complete success="true">john@example.com</complete>
577
577
  `}`;
578
578
  }