@anionex/dsh-vision-toolkit 0.1.18 → 0.1.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md
CHANGED
|
@@ -237,7 +237,7 @@ This is a shared zero-configuration entry point, not an unlimited private endpoi
|
|
|
237
237
|
| Images per request | Up to 5 |
|
|
238
238
|
| Image size | 4 MiB per image |
|
|
239
239
|
| Decoded pixels | 20,000,000 per image |
|
|
240
|
-
| Output |
|
|
240
|
+
| Output | Up to 4,096 tokens per request |
|
|
241
241
|
|
|
242
242
|
These safeguards prevent unusually large requests from monopolizing memory or request time. When shared capacity is reached, the service returns a readable `429` response with `Retry-After` instead of collapsing into an unexplained model failure.
|
|
243
243
|
|
package/README.zh.md
CHANGED
|
@@ -240,7 +240,7 @@ API Key: https://agent-vision.anionex.me(自动填写)
|
|
|
240
240
|
| 单次请求图片数 | 最多 5 张 |
|
|
241
241
|
| 单张图片大小 | 4 MiB |
|
|
242
242
|
| 单张图片像素 | 20,000,000 |
|
|
243
|
-
| 单次输出 |
|
|
243
|
+
| 单次输出 | 最多 4,096 tokens |
|
|
244
244
|
|
|
245
245
|
这些保护规则避免异常大的请求占满内存或请求时间。共享容量用尽时,服务会返回带 `Retry-After` 的明确 `429` 响应,不会只得到一个含糊的“模型失败”。
|
|
246
246
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@anionex/dsh-vision-toolkit",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.19",
|
|
4
4
|
"description": "DeepSeek Harness-native integration for agent-vision-toolkit: image Q&A, OCR, grounding, UI restoration, pixel diff, Artifacts, and Web UI.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"deepseek",
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
"repository": "https://github.com/Anionex/agent-vision-toolkit",
|
|
4
4
|
"version": "v0.1.0+snapshot.bc9803d",
|
|
5
5
|
"commit": "bc9803d7d6300c864d17460ecbb33540b26638e0",
|
|
6
|
-
"contentSha256": "
|
|
6
|
+
"contentSha256": "f730f247b0f3647187bd9f4ccf9e77cefd9b080489914bef81d2ba54a0e778e1",
|
|
7
7
|
"files": [
|
|
8
8
|
{
|
|
9
9
|
"path": "CHANGELOG.md",
|
|
@@ -47,13 +47,13 @@
|
|
|
47
47
|
},
|
|
48
48
|
{
|
|
49
49
|
"path": "detect.py",
|
|
50
|
-
"bytes":
|
|
51
|
-
"sha256": "
|
|
50
|
+
"bytes": 2218,
|
|
51
|
+
"sha256": "48a7070084f5b23b1477fa9a690ef1e679da8a03228e64a1ac583de633040bfc"
|
|
52
52
|
},
|
|
53
53
|
{
|
|
54
54
|
"path": "ground.py",
|
|
55
|
-
"bytes":
|
|
56
|
-
"sha256": "
|
|
55
|
+
"bytes": 10117,
|
|
56
|
+
"sha256": "845e56dbdf92f2c79495170f5d215de49985b3099012b4d398231d3c78bd0090"
|
|
57
57
|
},
|
|
58
58
|
{
|
|
59
59
|
"path": "skills/vision-tools/scripts/dominant_colors.py",
|
|
@@ -16,8 +16,12 @@ DEFAULT_CATEGORY = ("UI element (buttons, links, inputs, icons, labels, "
|
|
|
16
16
|
|
|
17
17
|
|
|
18
18
|
def build_target(category: str | None) -> str:
|
|
19
|
-
|
|
20
|
-
|
|
19
|
+
target = (category or DEFAULT_CATEGORY).strip()
|
|
20
|
+
if not target.lower().startswith("every distinct "):
|
|
21
|
+
target = f"every distinct {target}"
|
|
22
|
+
if "exact visible text" not in target.lower():
|
|
23
|
+
target += " — include the exact visible text in each label"
|
|
24
|
+
return target
|
|
21
25
|
|
|
22
26
|
|
|
23
27
|
def format_inventory(matches, width: int, height: int) -> list[str]:
|
|
@@ -4,6 +4,7 @@ import argparse
|
|
|
4
4
|
import base64
|
|
5
5
|
import io
|
|
6
6
|
import json
|
|
7
|
+
import os
|
|
7
8
|
import re
|
|
8
9
|
import sys
|
|
9
10
|
from dataclasses import dataclass
|
|
@@ -28,12 +29,23 @@ class GroundError(Exception):
|
|
|
28
29
|
pass
|
|
29
30
|
|
|
30
31
|
|
|
31
|
-
def
|
|
32
|
+
def coordinate_order() -> str:
|
|
33
|
+
configured = os.environ.get("VISION_BOX_ORDER", "").strip().lower()
|
|
34
|
+
if configured:
|
|
35
|
+
if configured not in {"xyxy", "yxyx"}:
|
|
36
|
+
raise GroundError("VISION_BOX_ORDER must be either xyxy or yxyx")
|
|
37
|
+
return configured
|
|
38
|
+
model = os.environ.get("VISION_MODEL", "").lower()
|
|
39
|
+
return "xyxy" if "qwen" in model else "yxyx"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def build_prompt(target: str, box_order: str = "yxyx") -> str:
|
|
43
|
+
coordinates = "[x0, y0, x1, y1]" if box_order == "xyxy" else "[y0, x0, y1, x1]"
|
|
32
44
|
return (
|
|
33
45
|
"Locate every visible object or region matching this target:\n"
|
|
34
46
|
f"{target}\n\n"
|
|
35
47
|
'Return only a JSON array. Each item must contain "box_2d" as '
|
|
36
|
-
'
|
|
48
|
+
f'{coordinates} on a 0-1000 grid and "label" as a short description. '
|
|
37
49
|
"Use tight boxes in the original image. Return [] when nothing matches."
|
|
38
50
|
)
|
|
39
51
|
|
|
@@ -70,6 +82,8 @@ def _items(text: str) -> list[Any]:
|
|
|
70
82
|
try:
|
|
71
83
|
payload = json.loads(cleaned)
|
|
72
84
|
except json.JSONDecodeError:
|
|
85
|
+
if cleaned.count("```") % 2 or _has_unclosed_json(cleaned):
|
|
86
|
+
raise GroundError("Vision API bounding-box JSON was truncated or incomplete") from None
|
|
73
87
|
fallback = _fallback_items(cleaned)
|
|
74
88
|
if fallback:
|
|
75
89
|
return fallback
|
|
@@ -83,7 +97,37 @@ def _items(text: str) -> list[Any]:
|
|
|
83
97
|
raise GroundError("Vision API returned an incompatible bounding-box JSON structure")
|
|
84
98
|
|
|
85
99
|
|
|
86
|
-
def
|
|
100
|
+
def _has_unclosed_json(text: str) -> bool:
|
|
101
|
+
start_positions = [position for position in (text.find("["), text.find("{")) if position >= 0]
|
|
102
|
+
if not start_positions:
|
|
103
|
+
return False
|
|
104
|
+
stack = []
|
|
105
|
+
in_string = False
|
|
106
|
+
escaped = False
|
|
107
|
+
pairs = {"]": "[", "}": "{"}
|
|
108
|
+
for character in text[min(start_positions):]:
|
|
109
|
+
if in_string:
|
|
110
|
+
if escaped:
|
|
111
|
+
escaped = False
|
|
112
|
+
elif character == "\\":
|
|
113
|
+
escaped = True
|
|
114
|
+
elif character == '"':
|
|
115
|
+
in_string = False
|
|
116
|
+
continue
|
|
117
|
+
if character == '"':
|
|
118
|
+
in_string = True
|
|
119
|
+
elif character in "[{":
|
|
120
|
+
stack.append(character)
|
|
121
|
+
elif character in "]}":
|
|
122
|
+
if not stack or stack[-1] != pairs[character]:
|
|
123
|
+
return False
|
|
124
|
+
stack.pop()
|
|
125
|
+
return bool(stack)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _normalize_box(
|
|
129
|
+
item: dict[str, Any], width: int, height: int, box_order: str = "yxyx",
|
|
130
|
+
) -> tuple[int, int, int, int] | None:
|
|
87
131
|
raw = item.get("box_2d")
|
|
88
132
|
if not isinstance(raw, list):
|
|
89
133
|
for key in ("bbox_2d", "box2d", "bbox", "box"):
|
|
@@ -93,9 +137,13 @@ def _normalize_box(item: dict[str, Any], width: int, height: int) -> tuple[int,
|
|
|
93
137
|
if not isinstance(raw, list) or len(raw) != 4:
|
|
94
138
|
return None
|
|
95
139
|
try:
|
|
96
|
-
|
|
140
|
+
values = tuple(float(value) for value in raw)
|
|
97
141
|
except (TypeError, ValueError):
|
|
98
142
|
return None
|
|
143
|
+
if box_order == "xyxy":
|
|
144
|
+
x0, y0, x1, y1 = values
|
|
145
|
+
else:
|
|
146
|
+
y0, x0, y1, x1 = values
|
|
99
147
|
if x0 > x1:
|
|
100
148
|
x0, x1 = x1, x0
|
|
101
149
|
if y0 > y1:
|
|
@@ -109,12 +157,14 @@ def _normalize_box(item: dict[str, Any], width: int, height: int) -> tuple[int,
|
|
|
109
157
|
return box if box[2] > box[0] and box[3] > box[1] else None
|
|
110
158
|
|
|
111
159
|
|
|
112
|
-
def parse_matches(
|
|
160
|
+
def parse_matches(
|
|
161
|
+
text: str, width: int, height: int, target: str, box_order: str = "yxyx",
|
|
162
|
+
) -> list[Match]:
|
|
113
163
|
matches = []
|
|
114
164
|
for item in _items(text):
|
|
115
165
|
if not isinstance(item, dict):
|
|
116
166
|
continue
|
|
117
|
-
box = _normalize_box(item, width, height)
|
|
167
|
+
box = _normalize_box(item, width, height, box_order)
|
|
118
168
|
if box is None:
|
|
119
169
|
continue
|
|
120
170
|
label = str(item.get("label") or item.get("caption") or item.get("description") or target).strip()
|
|
@@ -156,8 +206,9 @@ def locate(image_path: Path, target: str, region: str | None = None) -> list[Mat
|
|
|
156
206
|
width_used, height_used = box[2] - box[0], box[3] - box[1]
|
|
157
207
|
# 8192 leaves room for exhaustive targets ("every UI element"): a dense
|
|
158
208
|
# screen can emit dozens of boxes and 2048 truncated the JSON mid-array.
|
|
159
|
-
|
|
160
|
-
|
|
209
|
+
box_order = coordinate_order()
|
|
210
|
+
response = describe_image(url, build_prompt(target, box_order), max_tokens=8192)
|
|
211
|
+
matches = parse_matches(response, width_used, height_used, target, box_order)
|
|
161
212
|
if box is None:
|
|
162
213
|
return matches
|
|
163
214
|
# Matches were parsed in crop coordinates; report them in the original image.
|