@anionex/dsh-vision-toolkit 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. package/LICENSE +21 -0
  2. package/README.i18n.yaml +6 -0
  3. package/README.md +383 -0
  4. package/README.zh.md +383 -0
  5. package/assets/dsh-conversation-artifact.png +0 -0
  6. package/assets/dsh-conversation-image-qa-top.png +0 -0
  7. package/assets/dsh-conversation-image-qa.png +0 -0
  8. package/assets/dsh-conversation-pixel-diff.png +0 -0
  9. package/assets/dsh-conversation-screenshot-debugging-top.png +0 -0
  10. package/assets/dsh-conversation-screenshot-debugging.png +0 -0
  11. package/assets/dsh-conversation-tool-call.png +0 -0
  12. package/assets/dsh-conversation-vision-trace.png +0 -0
  13. package/assets/hero.png +0 -0
  14. package/assets/social-preview.png +0 -0
  15. package/assets/upstream/README.md +16 -0
  16. package/assets/upstream/image-qa.webp +0 -0
  17. package/assets/upstream/infographic-reference.webp +0 -0
  18. package/assets/upstream/infographic-result.webp +0 -0
  19. package/assets/upstream/screenshot-debugging.webp +0 -0
  20. package/assets/upstream/ui-result.webp +0 -0
  21. package/assets/upstream/ui-sketch.webp +0 -0
  22. package/assets/vision-settings.png +0 -0
  23. package/cordis.patch.yml +6 -0
  24. package/docs/assets/vision-settings.png +0 -0
  25. package/docs/requirements-traceability/README.i18n.yaml +6 -0
  26. package/docs/requirements-traceability/README.md +75 -0
  27. package/docs/requirements-traceability/README.zh.md +75 -0
  28. package/examples/ui-restoration/README.i18n.yaml +6 -0
  29. package/examples/ui-restoration/README.md +70 -0
  30. package/examples/ui-restoration/README.zh.md +70 -0
  31. package/examples/ui-restoration/assets/final-heatmap.png +0 -0
  32. package/examples/ui-restoration/assets/final-report.json +83 -0
  33. package/examples/ui-restoration/assets/implementation.png +0 -0
  34. package/examples/ui-restoration/assets/initial-heatmap.png +0 -0
  35. package/examples/ui-restoration/assets/initial-report.json +83 -0
  36. package/examples/ui-restoration/assets/initial.png +0 -0
  37. package/examples/ui-restoration/assets/metrics.json +12 -0
  38. package/examples/ui-restoration/assets/reference.png +0 -0
  39. package/examples/ui-restoration/implementation.html +94 -0
  40. package/examples/ui-restoration/initial.html +57 -0
  41. package/lib/artifact-access.js +369 -0
  42. package/lib/artifact-access.js.map +1 -0
  43. package/lib/artifacts.js +56 -0
  44. package/lib/artifacts.js.map +1 -0
  45. package/lib/client.js +952 -0
  46. package/lib/client.js.map +1 -0
  47. package/lib/config.js +117 -0
  48. package/lib/config.js.map +1 -0
  49. package/lib/errors.js +56 -0
  50. package/lib/errors.js.map +1 -0
  51. package/lib/exposure.js +213 -0
  52. package/lib/exposure.js.map +1 -0
  53. package/lib/index.js +97 -0
  54. package/lib/index.js.map +1 -0
  55. package/lib/paste-images.js +199 -0
  56. package/lib/paste-images.js.map +1 -0
  57. package/lib/paths.js +325 -0
  58. package/lib/paths.js.map +1 -0
  59. package/lib/runtime-install.js +601 -0
  60. package/lib/runtime-install.js.map +1 -0
  61. package/lib/runtime-manager.js +126 -0
  62. package/lib/runtime-manager.js.map +1 -0
  63. package/lib/runtime.js +1344 -0
  64. package/lib/runtime.js.map +1 -0
  65. package/lib/skill.js +139 -0
  66. package/lib/skill.js.map +1 -0
  67. package/lib/tools.js +528 -0
  68. package/lib/tools.js.map +1 -0
  69. package/lib/types/artifact-access.d.ts +61 -0
  70. package/lib/types/artifact-access.d.ts.map +1 -0
  71. package/lib/types/artifacts.d.ts +42 -0
  72. package/lib/types/artifacts.d.ts.map +1 -0
  73. package/lib/types/client/index.d.ts +179 -0
  74. package/lib/types/client/index.d.ts.map +1 -0
  75. package/lib/types/client/paste-images.d.ts +57 -0
  76. package/lib/types/client/paste-images.d.ts.map +1 -0
  77. package/lib/types/config.d.ts +73 -0
  78. package/lib/types/config.d.ts.map +1 -0
  79. package/lib/types/errors.d.ts +35 -0
  80. package/lib/types/errors.d.ts.map +1 -0
  81. package/lib/types/exposure.d.ts +40 -0
  82. package/lib/types/exposure.d.ts.map +1 -0
  83. package/lib/types/index.d.ts +18 -0
  84. package/lib/types/index.d.ts.map +1 -0
  85. package/lib/types/paste-images.d.ts +21 -0
  86. package/lib/types/paste-images.d.ts.map +1 -0
  87. package/lib/types/paths.d.ts +107 -0
  88. package/lib/types/paths.d.ts.map +1 -0
  89. package/lib/types/runtime-install.d.ts +49 -0
  90. package/lib/types/runtime-install.d.ts.map +1 -0
  91. package/lib/types/runtime-manager.d.ts +60 -0
  92. package/lib/types/runtime-manager.d.ts.map +1 -0
  93. package/lib/types/runtime.d.ts +389 -0
  94. package/lib/types/runtime.d.ts.map +1 -0
  95. package/lib/types/skill.d.ts +15 -0
  96. package/lib/types/skill.d.ts.map +1 -0
  97. package/lib/types/tools.d.ts +22 -0
  98. package/lib/types/tools.d.ts.map +1 -0
  99. package/lib/types/upstream.d.ts +207 -0
  100. package/lib/types/upstream.d.ts.map +1 -0
  101. package/lib/types/version.d.ts +15 -0
  102. package/lib/types/version.d.ts.map +1 -0
  103. package/lib/types/web-request.d.ts +4 -0
  104. package/lib/types/web-request.d.ts.map +1 -0
  105. package/lib/types/web.d.ts +74 -0
  106. package/lib/types/web.d.ts.map +1 -0
  107. package/lib/upstream.js +675 -0
  108. package/lib/upstream.js.map +1 -0
  109. package/lib/version.js +18 -0
  110. package/lib/version.js.map +1 -0
  111. package/lib/web-request.js +20 -0
  112. package/lib/web-request.js.map +1 -0
  113. package/lib/web.js +244 -0
  114. package/lib/web.js.map +1 -0
  115. package/package.json +139 -0
  116. package/runtime/requirements.lock +3 -0
  117. package/src/artifact-access.ts +386 -0
  118. package/src/artifacts.ts +85 -0
  119. package/src/client/index.tsx +866 -0
  120. package/src/client/paste-images.tsx +426 -0
  121. package/src/config.ts +177 -0
  122. package/src/errors.ts +62 -0
  123. package/src/exposure.ts +227 -0
  124. package/src/index.ts +122 -0
  125. package/src/paste-images.ts +234 -0
  126. package/src/paths.ts +348 -0
  127. package/src/runtime-install.ts +723 -0
  128. package/src/runtime-manager.ts +166 -0
  129. package/src/runtime.ts +1783 -0
  130. package/src/skill.ts +143 -0
  131. package/src/tools.ts +668 -0
  132. package/src/upstream.ts +861 -0
  133. package/src/version.ts +37 -0
  134. package/src/web-request.ts +17 -0
  135. package/src/web.ts +329 -0
  136. package/vendor/agent-vision-toolkit/CHANGELOG.md +16 -0
  137. package/vendor/agent-vision-toolkit/LICENSE +21 -0
  138. package/vendor/agent-vision-toolkit/README.md +399 -0
  139. package/vendor/agent-vision-toolkit/UPSTREAM_MANIFEST.json +89 -0
  140. package/vendor/agent-vision-toolkit/bin/crop +90 -0
  141. package/vendor/agent-vision-toolkit/bin/detect +13 -0
  142. package/vendor/agent-vision-toolkit/bin/glance +93 -0
  143. package/vendor/agent-vision-toolkit/bin/ground +13 -0
  144. package/vendor/agent-vision-toolkit/bin/trace +129 -0
  145. package/vendor/agent-vision-toolkit/detect.py +56 -0
  146. package/vendor/agent-vision-toolkit/ground.py +216 -0
  147. package/vendor/agent-vision-toolkit/skills/vision-tools/scripts/dominant_colors.py +224 -0
  148. package/vendor/agent-vision-toolkit/skills/vision-tools/scripts/extract_fg.py +278 -0
  149. package/vendor/agent-vision-toolkit/skills/vision-tools/scripts/html_shot.py +108 -0
  150. package/vendor/agent-vision-toolkit/skills/vision-tools/scripts/long_screenshot_ocr.py +1245 -0
  151. package/vendor/agent-vision-toolkit/skills/vision-tools/scripts/pixel_diff.py +88 -0
  152. package/vendor/agent-vision-toolkit/vision_client.py +156 -0
@@ -0,0 +1,88 @@
1
+ #!/usr/bin/env python3
2
+ """pixel_diff: compare a rebuilt image against the original and rank the worst regions.
3
+
4
+ Verification is the step agents skip or fake, because comparing two
5
+ descriptions feels like comparing two images. It is not. This runs the real
6
+ comparison and answers the only question that matters next: which region is
7
+ most wrong, so you know where to look and what to fix first.
8
+
9
+ Boxes are printed in the same `x1: .., y1: ..` form as ground/detect, so a
10
+ bad region can be pasted straight into `glance --region` or `detect --region`.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import argparse
16
+ from pathlib import Path
17
+
18
+ try:
19
+ from PIL import Image, ImageChops, ImageStat
20
+ except ImportError:
21
+ Image = None
22
+
23
+
24
+ def load(path: Path, size: tuple[int, int] | None = None) -> "Image.Image":
25
+ # Traced SVGs and screenshots of transparent UI render with an alpha
26
+ # channel; unflattened transparency reads as black and shows up as a huge
27
+ # phantom diff. Compositing on white is what a viewer would show.
28
+ image = Image.open(path)
29
+ if image.mode in ("RGBA", "LA", "P"):
30
+ image = image.convert("RGBA")
31
+ canvas = Image.new("RGB", image.size, "white")
32
+ canvas.paste(image, mask=image.split()[-1])
33
+ image = canvas
34
+ else:
35
+ image = image.convert("RGB")
36
+ return image.resize(size, Image.LANCZOS) if size and image.size != size else image
37
+
38
+
39
+ def cell_scores(diff: "Image.Image", grid: int) -> list[tuple[float, tuple[int, int, int, int]]]:
40
+ width, height = diff.size
41
+ grey = diff.convert("L")
42
+ scores = []
43
+ for row in range(grid):
44
+ for column in range(grid):
45
+ box = (round(column * width / grid), round(row * height / grid),
46
+ round((column + 1) * width / grid), round((row + 1) * height / grid))
47
+ if box[2] > box[0] and box[3] > box[1]:
48
+ mean = ImageStat.Stat(grey.crop(box)).mean[0]
49
+ scores.append((mean / 255 * 100, box))
50
+ return sorted(scores, reverse=True)
51
+
52
+
53
+ def main() -> None:
54
+ parser = argparse.ArgumentParser(
55
+ prog="pixel_diff",
56
+ description="Pixel-diff a rebuilt image against the original and rank the worst regions",
57
+ )
58
+ parser.add_argument("original", type=Path, help="the reference image")
59
+ parser.add_argument("rebuilt", type=Path, help="your rendered reproduction")
60
+ parser.add_argument("--grid", type=int, default=6, help="split into GRID x GRID cells (default: 6)")
61
+ parser.add_argument("--top", type=int, default=5, help="how many worst regions to print (default: 5)")
62
+ parser.add_argument("-o", "--output", type=Path, help="write a diff heatmap image here")
63
+ args = parser.parse_args()
64
+ if Image is None:
65
+ parser.exit(1, "pixel_diff: requires Pillow; install the optional dependency pillow first\n")
66
+ for path in (args.original, args.rebuilt):
67
+ if not path.expanduser().is_file():
68
+ parser.exit(1, f"pixel_diff: image not found: {path}\n")
69
+ original = load(args.original.expanduser())
70
+ with Image.open(args.rebuilt.expanduser()) as probe:
71
+ raw_size = probe.size
72
+ rebuilt = load(args.rebuilt.expanduser(), size=original.size)
73
+ if raw_size != original.size:
74
+ # Worth saying out loud: a size mismatch is itself a finding, and every
75
+ # box printed below is in the original's coordinates, not the rebuild's.
76
+ print(f"note: rebuilt was {raw_size[0]}x{raw_size[1]}, scaled to {original.size[0]}x{original.size[1]}")
77
+ diff = ImageChops.difference(original, rebuilt)
78
+ overall = ImageStat.Stat(diff.convert("L")).mean[0] / 255 * 100
79
+ print(f"overall difference: {overall:.2f}%")
80
+ if args.output:
81
+ diff.save(args.output.expanduser())
82
+ print(f"heatmap: {args.output}")
83
+ for index, (score, box) in enumerate(cell_scores(diff, args.grid)[:args.top], 1):
84
+ print(f"{index}. {score:.2f}% x1: {box[0]}, y1: {box[1]}, x2: {box[2]}, y2: {box[3]}")
85
+
86
+
87
+ if __name__ == "__main__":
88
+ main()
@@ -0,0 +1,156 @@
1
+ #!/usr/bin/env python3
2
+ """Shared OpenAI-compatible vision client used by the proxy and glance CLI."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import base64
7
+ import http.client
8
+ import json
9
+ import mimetypes
10
+ import os
11
+ from pathlib import Path
12
+ import sys
13
+ import time
14
+ import urllib.error
15
+ import urllib.request
16
+
17
+ DEFAULT_PROMPT = "Please describe the contents of this image in detail."
18
+
19
+ LANG_INSTRUCTIONS = {
20
+ "zh": "请使用简体中文回答。",
21
+ "en": "Please respond in English.",
22
+ }
23
+
24
+
25
+ class VisionError(RuntimeError):
26
+ """A safe, user-facing vision request failure."""
27
+
28
+
29
+ def load_env_file(path: str | os.PathLike[str] | None) -> None:
30
+ if not path:
31
+ return
32
+ env_path = Path(path).expanduser()
33
+ if not env_path.is_file():
34
+ return
35
+ for raw_line in env_path.read_text().splitlines():
36
+ line = raw_line.strip()
37
+ if not line or line.startswith("#") or "=" not in line:
38
+ continue
39
+ key, _, value = line.partition("=")
40
+ key = key.strip()
41
+ value = value.strip().strip('"').strip("'")
42
+ # The env file is the user's explicit configuration: whatever it sets wins,
43
+ # even when the same variable already exists in the system environment.
44
+ if key:
45
+ os.environ[key] = value
46
+
47
+
48
+ def load_default_env() -> None:
49
+ explicit = os.environ.get("VISION_ENV_FILE")
50
+ candidates = [Path(explicit).expanduser()] if explicit else []
51
+ local_appdata = os.environ.get("LOCALAPPDATA")
52
+ if local_appdata:
53
+ candidates.append(Path(local_appdata) / "agent-vision-toolkit" / "env")
54
+ candidates.extend([
55
+ Path.home() / ".config" / "agent-vision-toolkit" / "env",
56
+ Path(__file__).resolve().parent / ".env",
57
+ Path.cwd() / ".env",
58
+ ])
59
+ for path in candidates:
60
+ load_env_file(path)
61
+
62
+
63
+ def _required(name: str) -> str:
64
+ value = os.environ.get(name, "").strip()
65
+ if not value:
66
+ raise VisionError(f"Missing config {name}; fill it in the .env file")
67
+ return value
68
+
69
+
70
+ def validate_vision_config() -> None:
71
+ for name in ("VISION_API_KEY", "VISION_BASE_URL", "VISION_MODEL"):
72
+ _required(name)
73
+
74
+
75
+ def image_path_to_data_url(path: str | os.PathLike[str]) -> str:
76
+ image_path = Path(path).expanduser()
77
+ if not image_path.is_file():
78
+ raise VisionError(f"Image not found: {image_path}")
79
+ mime, _ = mimetypes.guess_type(image_path.name)
80
+ if mime not in {"image/png", "image/jpeg", "image/gif", "image/webp"}:
81
+ raise VisionError("Only PNG, JPEG, GIF, and WebP images are supported")
82
+ return f"data:{mime};base64,{base64.b64encode(image_path.read_bytes()).decode()}"
83
+
84
+
85
+ def _message_text(message: object) -> str:
86
+ if isinstance(message, str):
87
+ return message.strip()
88
+ if isinstance(message, list):
89
+ return "\n".join(
90
+ part["text"] for part in message
91
+ if isinstance(part, dict) and isinstance(part.get("text"), str)
92
+ ).strip()
93
+ return ""
94
+
95
+
96
+ def describe_image(image_url: str | list[str], prompt: str | None = None, max_tokens: int = 4096,
97
+ apply_lang: bool = True) -> str:
98
+ """Describe one data/http image URL (str) or several (list) in a single call."""
99
+ validate_vision_config()
100
+ urls = [image_url] if isinstance(image_url, str) else list(image_url)
101
+ if not urls:
102
+ raise VisionError("No image was provided")
103
+ for url in urls:
104
+ if not url.startswith(("data:", "http://", "https://")):
105
+ raise VisionError("Only data URLs or http(s) image URLs are supported")
106
+ base_url = _required("VISION_BASE_URL").rstrip("/")
107
+ api_key = _required("VISION_API_KEY")
108
+ text = prompt or DEFAULT_PROMPT
109
+ if apply_lang:
110
+ instruction = LANG_INSTRUCTIONS.get(os.environ.get("LANG", "").strip().lower())
111
+ if instruction:
112
+ text = f"{instruction}\n\n{text}"
113
+ payload = {
114
+ "model": _required("VISION_MODEL"),
115
+ "messages": [{"role": "user", "content": [{"type": "text", "text": text}] + [
116
+ {"type": "image_url", "image_url": {"url": url}} for url in urls
117
+ ]}],
118
+ }
119
+ if max_tokens is not None:
120
+ payload["max_tokens"] = max_tokens
121
+ request = urllib.request.Request(
122
+ base_url + "/chat/completions",
123
+ data=json.dumps(payload).encode(),
124
+ headers={"Content-Type": "application/json", "Authorization": "Bearer " + api_key},
125
+ )
126
+ retries = 2
127
+ timeout = 180
128
+ for attempt in range(retries + 1):
129
+ try:
130
+ with urllib.request.urlopen(request, timeout=timeout) as response:
131
+ data = json.load(response)
132
+ try:
133
+ text = _message_text(data["choices"][0]["message"]["content"])
134
+ except (KeyError, IndexError, TypeError) as exc:
135
+ raise VisionError("Vision API returned an incompatible response structure") from exc
136
+ if not text:
137
+ raise VisionError("Vision API returned an empty description")
138
+ return text
139
+ except urllib.error.HTTPError as exc:
140
+ body = exc.read().decode(errors="replace")[:400].replace(api_key, "<redacted>")
141
+ body = body.replace("\r", " ").replace("\n", " ")
142
+ if exc.code in {429, 500, 502, 503, 504} and attempt < retries:
143
+ print(f"vision: HTTP {exc.code}, retrying ({attempt + 1}/{retries})", file=sys.stderr)
144
+ time.sleep(min(2 ** attempt, 4))
145
+ continue
146
+ raise VisionError(f"Vision API HTTP {exc.code}: {body}") from exc
147
+ except (urllib.error.URLError, TimeoutError, ConnectionError, http.client.IncompleteRead) as exc:
148
+ if attempt < retries:
149
+ print(f"vision: {type(exc).__name__}, retrying ({attempt + 1}/{retries})", file=sys.stderr)
150
+ time.sleep(min(2 ** attempt, 4))
151
+ continue
152
+ reason = getattr(exc, "reason", str(exc))
153
+ raise VisionError(f"Vision API network error: {reason}") from exc
154
+ except json.JSONDecodeError as exc:
155
+ raise VisionError("Vision API returned invalid JSON") from exc
156
+ raise VisionError("Vision API request failed")