@anionex/dsh-vision-toolkit 0.1.18 → 0.1.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -237,7 +237,7 @@ This is a shared zero-configuration entry point, not an unlimited private endpoi
237
237
  | Images per request | Up to 5 |
238
238
  | Image size | 4 MiB per image |
239
239
  | Decoded pixels | 20,000,000 per image |
240
- | Output | 512 tokens per request |
240
+ | Output | Up to 4,096 tokens per request |
241
241
 
242
242
  These safeguards prevent unusually large requests from monopolizing memory or request time. When shared capacity is reached, the service returns a readable `429` response with `Retry-After` instead of collapsing into an unexplained model failure.
243
243
 
package/README.zh.md CHANGED
@@ -240,7 +240,7 @@ API Key: https://agent-vision.anionex.me(自动填写)
240
240
  | 单次请求图片数 | 最多 5 张 |
241
241
  | 单张图片大小 | 4 MiB |
242
242
  | 单张图片像素 | 20,000,000 |
243
- | 单次输出 | 512 tokens |
243
+ | 单次输出 | 最多 4,096 tokens |
244
244
 
245
245
  这些保护规则避免异常大的请求占满内存或请求时间。共享容量用尽时,服务会返回带 `Retry-After` 的明确 `429` 响应,不会只得到一个含糊的“模型失败”。
246
246
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@anionex/dsh-vision-toolkit",
3
- "version": "0.1.18",
3
+ "version": "0.1.19",
4
4
  "description": "DeepSeek Harness-native integration for agent-vision-toolkit: image Q&A, OCR, grounding, UI restoration, pixel diff, Artifacts, and Web UI.",
5
5
  "keywords": [
6
6
  "deepseek",
@@ -3,7 +3,7 @@
3
3
  "repository": "https://github.com/Anionex/agent-vision-toolkit",
4
4
  "version": "v0.1.0+snapshot.bc9803d",
5
5
  "commit": "bc9803d7d6300c864d17460ecbb33540b26638e0",
6
- "contentSha256": "dbcc6d6214976be415ee660557d4ad0abb8b59b91a9d9ca07ef5fda8fd9cf4b9",
6
+ "contentSha256": "f730f247b0f3647187bd9f4ccf9e77cefd9b080489914bef81d2ba54a0e778e1",
7
7
  "files": [
8
8
  {
9
9
  "path": "CHANGELOG.md",
@@ -47,13 +47,13 @@
47
47
  },
48
48
  {
49
49
  "path": "detect.py",
50
- "bytes": 2052,
51
- "sha256": "0e79f6cfbeb27d52cea57a6a642bb658d9c9d3f10e175416c320d91affc1d562"
50
+ "bytes": 2218,
51
+ "sha256": "48a7070084f5b23b1477fa9a690ef1e679da8a03228e64a1ac583de633040bfc"
52
52
  },
53
53
  {
54
54
  "path": "ground.py",
55
- "bytes": 8361,
56
- "sha256": "52c1b69f3c78f37acd33f7f542bdbceeabb3546e8e28be7278d72e04024699b9"
55
+ "bytes": 10117,
56
+ "sha256": "845e56dbdf92f2c79495170f5d215de49985b3099012b4d398231d3c78bd0090"
57
57
  },
58
58
  {
59
59
  "path": "skills/vision-tools/scripts/dominant_colors.py",
@@ -16,8 +16,12 @@ DEFAULT_CATEGORY = ("UI element (buttons, links, inputs, icons, labels, "
16
16
 
17
17
 
18
18
  def build_target(category: str | None) -> str:
19
- return (f"every distinct {category or DEFAULT_CATEGORY} — "
20
- "include the exact visible text in each label")
19
+ target = (category or DEFAULT_CATEGORY).strip()
20
+ if not target.lower().startswith("every distinct "):
21
+ target = f"every distinct {target}"
22
+ if "exact visible text" not in target.lower():
23
+ target += " — include the exact visible text in each label"
24
+ return target
21
25
 
22
26
 
23
27
  def format_inventory(matches, width: int, height: int) -> list[str]:
@@ -4,6 +4,7 @@ import argparse
4
4
  import base64
5
5
  import io
6
6
  import json
7
+ import os
7
8
  import re
8
9
  import sys
9
10
  from dataclasses import dataclass
@@ -28,12 +29,23 @@ class GroundError(Exception):
28
29
  pass
29
30
 
30
31
 
31
- def build_prompt(target: str) -> str:
32
+ def coordinate_order() -> str:
33
+ configured = os.environ.get("VISION_BOX_ORDER", "").strip().lower()
34
+ if configured:
35
+ if configured not in {"xyxy", "yxyx"}:
36
+ raise GroundError("VISION_BOX_ORDER must be either xyxy or yxyx")
37
+ return configured
38
+ model = os.environ.get("VISION_MODEL", "").lower()
39
+ return "xyxy" if "qwen" in model else "yxyx"
40
+
41
+
42
+ def build_prompt(target: str, box_order: str = "yxyx") -> str:
43
+ coordinates = "[x0, y0, x1, y1]" if box_order == "xyxy" else "[y0, x0, y1, x1]"
32
44
  return (
33
45
  "Locate every visible object or region matching this target:\n"
34
46
  f"{target}\n\n"
35
47
  'Return only a JSON array. Each item must contain "box_2d" as '
36
- '[y0, x0, y1, x1] on a 0-1000 grid and "label" as a short description. '
48
+ f'{coordinates} on a 0-1000 grid and "label" as a short description. '
37
49
  "Use tight boxes in the original image. Return [] when nothing matches."
38
50
  )
39
51
 
@@ -70,6 +82,8 @@ def _items(text: str) -> list[Any]:
70
82
  try:
71
83
  payload = json.loads(cleaned)
72
84
  except json.JSONDecodeError:
85
+ if cleaned.count("```") % 2 or _has_unclosed_json(cleaned):
86
+ raise GroundError("Vision API bounding-box JSON was truncated or incomplete") from None
73
87
  fallback = _fallback_items(cleaned)
74
88
  if fallback:
75
89
  return fallback
@@ -83,7 +97,37 @@ def _items(text: str) -> list[Any]:
83
97
  raise GroundError("Vision API returned an incompatible bounding-box JSON structure")
84
98
 
85
99
 
86
- def _normalize_box(item: dict[str, Any], width: int, height: int) -> tuple[int, int, int, int] | None:
100
+ def _has_unclosed_json(text: str) -> bool:
101
+ start_positions = [position for position in (text.find("["), text.find("{")) if position >= 0]
102
+ if not start_positions:
103
+ return False
104
+ stack = []
105
+ in_string = False
106
+ escaped = False
107
+ pairs = {"]": "[", "}": "{"}
108
+ for character in text[min(start_positions):]:
109
+ if in_string:
110
+ if escaped:
111
+ escaped = False
112
+ elif character == "\\":
113
+ escaped = True
114
+ elif character == '"':
115
+ in_string = False
116
+ continue
117
+ if character == '"':
118
+ in_string = True
119
+ elif character in "[{":
120
+ stack.append(character)
121
+ elif character in "]}":
122
+ if not stack or stack[-1] != pairs[character]:
123
+ return False
124
+ stack.pop()
125
+ return bool(stack)
126
+
127
+
128
+ def _normalize_box(
129
+ item: dict[str, Any], width: int, height: int, box_order: str = "yxyx",
130
+ ) -> tuple[int, int, int, int] | None:
87
131
  raw = item.get("box_2d")
88
132
  if not isinstance(raw, list):
89
133
  for key in ("bbox_2d", "box2d", "bbox", "box"):
@@ -93,9 +137,13 @@ def _normalize_box(item: dict[str, Any], width: int, height: int) -> tuple[int,
93
137
  if not isinstance(raw, list) or len(raw) != 4:
94
138
  return None
95
139
  try:
96
- y0, x0, y1, x1 = (float(value) for value in raw)
140
+ values = tuple(float(value) for value in raw)
97
141
  except (TypeError, ValueError):
98
142
  return None
143
+ if box_order == "xyxy":
144
+ x0, y0, x1, y1 = values
145
+ else:
146
+ y0, x0, y1, x1 = values
99
147
  if x0 > x1:
100
148
  x0, x1 = x1, x0
101
149
  if y0 > y1:
@@ -109,12 +157,14 @@ def _normalize_box(item: dict[str, Any], width: int, height: int) -> tuple[int,
109
157
  return box if box[2] > box[0] and box[3] > box[1] else None
110
158
 
111
159
 
112
- def parse_matches(text: str, width: int, height: int, target: str) -> list[Match]:
160
+ def parse_matches(
161
+ text: str, width: int, height: int, target: str, box_order: str = "yxyx",
162
+ ) -> list[Match]:
113
163
  matches = []
114
164
  for item in _items(text):
115
165
  if not isinstance(item, dict):
116
166
  continue
117
- box = _normalize_box(item, width, height)
167
+ box = _normalize_box(item, width, height, box_order)
118
168
  if box is None:
119
169
  continue
120
170
  label = str(item.get("label") or item.get("caption") or item.get("description") or target).strip()
@@ -156,8 +206,9 @@ def locate(image_path: Path, target: str, region: str | None = None) -> list[Mat
156
206
  width_used, height_used = box[2] - box[0], box[3] - box[1]
157
207
  # 8192 leaves room for exhaustive targets ("every UI element"): a dense
158
208
  # screen can emit dozens of boxes and 2048 truncated the JSON mid-array.
159
- response = describe_image(url, build_prompt(target), max_tokens=8192)
160
- matches = parse_matches(response, width_used, height_used, target)
209
+ box_order = coordinate_order()
210
+ response = describe_image(url, build_prompt(target, box_order), max_tokens=8192)
211
+ matches = parse_matches(response, width_used, height_used, target, box_order)
161
212
  if box is None:
162
213
  return matches
163
214
  # Matches were parsed in crop coordinates; report them in the original image.