xyberos-vision 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- xyberos_vision-0.1.0/PKG-INFO +62 -0
- xyberos_vision-0.1.0/README.md +47 -0
- xyberos_vision-0.1.0/pyproject.toml +27 -0
- xyberos_vision-0.1.0/setup.cfg +4 -0
- xyberos_vision-0.1.0/tests/test_adapters.py +56 -0
- xyberos_vision-0.1.0/tests/test_plugin.py +32 -0
- xyberos_vision-0.1.0/xyberos_vision/__init__.py +15 -0
- xyberos_vision-0.1.0/xyberos_vision/adapters.py +140 -0
- xyberos_vision-0.1.0/xyberos_vision/contract.py +38 -0
- xyberos_vision-0.1.0/xyberos_vision/http.py +90 -0
- xyberos_vision-0.1.0/xyberos_vision/plugin.py +85 -0
- xyberos_vision-0.1.0/xyberos_vision.egg-info/PKG-INFO +62 -0
- xyberos_vision-0.1.0/xyberos_vision.egg-info/SOURCES.txt +15 -0
- xyberos_vision-0.1.0/xyberos_vision.egg-info/dependency_links.txt +1 -0
- xyberos_vision-0.1.0/xyberos_vision.egg-info/entry_points.txt +2 -0
- xyberos_vision-0.1.0/xyberos_vision.egg-info/requires.txt +8 -0
- xyberos_vision-0.1.0/xyberos_vision.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: xyberos-vision
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Vision plugin (RFC-0019, M8): describe images, OCR, generate images
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Keywords: xyberos,plugin,vision,ocr,tesseract,image
|
|
7
|
+
Requires-Python: >=3.10
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
Requires-Dist: xyberos>=1.0
|
|
10
|
+
Provides-Extra: ocr
|
|
11
|
+
Requires-Dist: pytesseract; extra == "ocr"
|
|
12
|
+
Requires-Dist: pillow; extra == "ocr"
|
|
13
|
+
Provides-Extra: test
|
|
14
|
+
Requires-Dist: pytest; extra == "test"
|
|
15
|
+
|
|
16
|
+
# xyberos-vision
|
|
17
|
+
|
|
18
|
+
**Vision plugin — RFC-0019, M8.** Describe images (vision model), extract text
|
|
19
|
+
(OCR), and generate images — three tools, three contracts.
|
|
20
|
+
|
|
21
|
+
## Install
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install -e ./vision
|
|
25
|
+
pip install xyberos-vision[ocr] # optional: pytesseract + pillow for OCR
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Usage
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from xyberos import create_app
|
|
32
|
+
from xyberos_vision import VisionPlugin
|
|
33
|
+
|
|
34
|
+
app = create_app()
|
|
35
|
+
app.load_plugin(VisionPlugin()) # key from OPENAI_API_KEY
|
|
36
|
+
|
|
37
|
+
app.tools.execute("vision_describe", None, image_path="photo.png", prompt="What's in this photo?")
|
|
38
|
+
app.tools.execute("ocr_extract_text", None, image_path="scan.png")
|
|
39
|
+
app.tools.execute("image_generate", None, prompt="a sunset over the ocean", output_path="out.png")
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Tools
|
|
43
|
+
|
|
44
|
+
| Tool | Backed by |
|
|
45
|
+
| ---- | --------- |
|
|
46
|
+
| `vision_describe(image_path, prompt=...)` | `OpenAICompatibleVision` (`VISION_MODEL`, default `gpt-4o-mini`) |
|
|
47
|
+
| `ocr_extract_text(image_path)` | `TesseractOCR` (lazy `pytesseract` + `pillow`) |
|
|
48
|
+
| `image_generate(prompt, output_path)` | `OpenAIImageGenerator` (`dall-e-3`) |
|
|
49
|
+
|
|
50
|
+
## Tests
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
pip install pytest
|
|
54
|
+
pytest tests/
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Vision/image-gen are tested via injectable transports; OCR tests skip when
|
|
58
|
+
`pytesseract` is absent.
|
|
59
|
+
|
|
60
|
+
## Ship location
|
|
61
|
+
|
|
62
|
+
Plugin (`xyberos.plugins` entry point) — vision/multimodal (M8).
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# xyberos-vision
|
|
2
|
+
|
|
3
|
+
**Vision plugin — RFC-0019, M8.** Describe images (vision model), extract text
|
|
4
|
+
(OCR), and generate images — three tools, three contracts.
|
|
5
|
+
|
|
6
|
+
## Install
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
pip install -e ./vision
|
|
10
|
+
pip install xyberos-vision[ocr] # optional: pytesseract + pillow for OCR
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Usage
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
from xyberos import create_app
|
|
17
|
+
from xyberos_vision import VisionPlugin
|
|
18
|
+
|
|
19
|
+
app = create_app()
|
|
20
|
+
app.load_plugin(VisionPlugin()) # key from OPENAI_API_KEY
|
|
21
|
+
|
|
22
|
+
app.tools.execute("vision_describe", None, image_path="photo.png", prompt="What's in this photo?")
|
|
23
|
+
app.tools.execute("ocr_extract_text", None, image_path="scan.png")
|
|
24
|
+
app.tools.execute("image_generate", None, prompt="a sunset over the ocean", output_path="out.png")
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
## Tools
|
|
28
|
+
|
|
29
|
+
| Tool | Backed by |
|
|
30
|
+
| ---- | --------- |
|
|
31
|
+
| `vision_describe(image_path, prompt=...)` | `OpenAICompatibleVision` (`VISION_MODEL`, default `gpt-4o-mini`) |
|
|
32
|
+
| `ocr_extract_text(image_path)` | `TesseractOCR` (lazy `pytesseract` + `pillow`) |
|
|
33
|
+
| `image_generate(prompt, output_path)` | `OpenAIImageGenerator` (`dall-e-3`) |
|
|
34
|
+
|
|
35
|
+
## Tests
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install pytest
|
|
39
|
+
pytest tests/
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Vision/image-gen are tested via injectable transports; OCR tests skip when
|
|
43
|
+
`pytesseract` is absent.
|
|
44
|
+
|
|
45
|
+
## Ship location
|
|
46
|
+
|
|
47
|
+
Plugin (`xyberos.plugins` entry point) — vision/multimodal (M8).
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61.0"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "xyberos-vision"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Vision plugin (RFC-0019, M8): describe images, OCR, generate images"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = {text = "Apache-2.0"}
|
|
12
|
+
dependencies = ["xyberos>=1.0"]
|
|
13
|
+
keywords = ["xyberos", "plugin", "vision", "ocr", "tesseract", "image"]
|
|
14
|
+
|
|
15
|
+
[project.optional-dependencies]
|
|
16
|
+
ocr = ["pytesseract", "pillow"]
|
|
17
|
+
test = ["pytest"]
|
|
18
|
+
|
|
19
|
+
[project.entry-points."xyberos.plugins"]
|
|
20
|
+
vision = "xyberos_vision.plugin:plugin"
|
|
21
|
+
|
|
22
|
+
[tool.setuptools]
|
|
23
|
+
packages = ["xyberos_vision"]
|
|
24
|
+
|
|
25
|
+
[tool.pytest.ini_options]
|
|
26
|
+
testpaths = ["tests"]
|
|
27
|
+
pythonpath = ["."]
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Tests for the vision adapters (injectable transports / skips, no network)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import base64
|
|
6
|
+
import importlib.util
|
|
7
|
+
|
|
8
|
+
import pytest
|
|
9
|
+
from xyberos.exceptions.provider import ProviderError
|
|
10
|
+
|
|
11
|
+
from xyberos_vision import OpenAICompatibleVision, OpenAIImageGenerator, TesseractOCR
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _tiny_png(tmp_path) -> str:
|
|
15
|
+
# A 1x1 PNG so image reading + base64 works without real image deps.
|
|
16
|
+
path = tmp_path / "img.png"
|
|
17
|
+
path.write_bytes(b"\x89PNG\r\n\x1a\n\x00\x00\x00\rioEND")
|
|
18
|
+
return str(path)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_vision_describe(tmp_path):
|
|
22
|
+
captured = {}
|
|
23
|
+
|
|
24
|
+
def request(method, url, **kwargs):
|
|
25
|
+
captured.update(kwargs)
|
|
26
|
+
return 200, {"choices": [{"message": {"content": "A cat"}}]}
|
|
27
|
+
|
|
28
|
+
image = _tiny_png(tmp_path)
|
|
29
|
+
vision = OpenAICompatibleVision(api_key="k", request=request)
|
|
30
|
+
assert vision.describe(image, "What is this?") == "A cat"
|
|
31
|
+
content = captured["json_body"]["messages"][0]["content"]
|
|
32
|
+
assert content[0]["text"] == "What is this?"
|
|
33
|
+
assert content[1]["type"] == "image_url"
|
|
34
|
+
assert content[1]["image_url"]["url"].startswith("data:image/png;base64,")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def test_vision_requires_key():
|
|
38
|
+
with pytest.raises(ProviderError, match="OPENAI_API_KEY"):
|
|
39
|
+
OpenAICompatibleVision(api_key=None, request=lambda *a, **k: (200, {})).describe("x.png", "p")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def test_image_generate(tmp_path):
|
|
43
|
+
def request(method, url, **kwargs):
|
|
44
|
+
return 200, {"data": [{"b64_json": base64.b64encode(b"PNGDATA").decode("ascii")}]}
|
|
45
|
+
|
|
46
|
+
output = tmp_path / "gen.png"
|
|
47
|
+
gen = OpenAIImageGenerator(api_key="k", request=request)
|
|
48
|
+
assert gen.generate("a dog", str(output)) == str(output)
|
|
49
|
+
assert output.read_bytes() == b"PNGDATA"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def test_ocr():
|
|
53
|
+
if importlib.util.find_spec("pytesseract") is None:
|
|
54
|
+
pytest.skip("pytesseract is not installed")
|
|
55
|
+
ocr = TesseractOCR()
|
|
56
|
+
assert isinstance(ocr, TesseractOCR)
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Tests for loading the vision plugin into a Xyberos app."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import base64
|
|
6
|
+
|
|
7
|
+
from xyberos import create_app
|
|
8
|
+
|
|
9
|
+
from xyberos_vision import VisionPlugin
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def test_plugin_registers_tools(tmp_path):
|
|
13
|
+
def request(method, url, **kwargs):
|
|
14
|
+
if url.endswith("/images/generations"):
|
|
15
|
+
return 200, {"data": [{"b64_json": base64.b64encode(b"PNG").decode("ascii")}]}
|
|
16
|
+
return 200, {"choices": [{"message": {"content": "A cat"}}]}
|
|
17
|
+
|
|
18
|
+
image = tmp_path / "img.png"
|
|
19
|
+
image.write_bytes(b"\x89PNG\r\n\x1a\n\x00\x00\x00\rioEND")
|
|
20
|
+
|
|
21
|
+
app = create_app()
|
|
22
|
+
app.load_plugin(VisionPlugin(api_key="k", request=request))
|
|
23
|
+
assert "vision_describe" in app.tools.names
|
|
24
|
+
assert "ocr_extract_text" in app.tools.names
|
|
25
|
+
assert "image_generate" in app.tools.names
|
|
26
|
+
|
|
27
|
+
assert app.tools.execute("vision_describe", None, image_path=str(image)) == "A cat"
|
|
28
|
+
output = tmp_path / "gen.png"
|
|
29
|
+
assert app.tools.execute("image_generate", None, prompt="a dog", output_path=str(output)) == str(output)
|
|
30
|
+
assert output.read_bytes() == b"PNG"
|
|
31
|
+
|
|
32
|
+
app.unload_plugin("vision")
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Vision plugin (RFC-0019, M8)."""
|
|
2
|
+
|
|
3
|
+
from .adapters import OpenAICompatibleVision, OpenAIImageGenerator, TesseractOCR
|
|
4
|
+
from .contract import ImageGenerator, OCR, VisionModel
|
|
5
|
+
from .plugin import VisionPlugin
|
|
6
|
+
|
|
7
|
+
__all__ = [
|
|
8
|
+
"ImageGenerator",
|
|
9
|
+
"OCR",
|
|
10
|
+
"OpenAICompatibleVision",
|
|
11
|
+
"OpenAIImageGenerator",
|
|
12
|
+
"TesseractOCR",
|
|
13
|
+
"VisionModel",
|
|
14
|
+
"VisionPlugin",
|
|
15
|
+
]
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""Vision adapters: OpenAI-compatible vision, Tesseract OCR, image generation."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import base64
|
|
6
|
+
import importlib
|
|
7
|
+
import mimetypes
|
|
8
|
+
import os
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from xyberos.exceptions.provider import ProviderError
|
|
13
|
+
|
|
14
|
+
from .http import RequestTransport, default_request
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _raise_for_status(status: int, body: Any) -> None:
|
|
18
|
+
if 200 <= status < 300:
|
|
19
|
+
return
|
|
20
|
+
message = body if isinstance(body, str) else str(body)
|
|
21
|
+
raise ProviderError(f"vision API returned HTTP {status}: {message[:200]}")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _image_data_url(image_path: str) -> str:
|
|
25
|
+
path = Path(image_path)
|
|
26
|
+
if not path.is_file():
|
|
27
|
+
raise FileNotFoundError(image_path)
|
|
28
|
+
mime = mimetypes.guess_type(str(path))[0] or "image/png"
|
|
29
|
+
encoded = base64.b64encode(path.read_bytes()).decode("ascii")
|
|
30
|
+
return f"data:{mime};base64,{encoded}"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class OpenAICompatibleVision:
|
|
34
|
+
"""Chat-completions vision model (works with OpenAI / Gemini-compatible)."""
|
|
35
|
+
|
|
36
|
+
name = "openai_compatible"
|
|
37
|
+
|
|
38
|
+
def __init__(
|
|
39
|
+
self,
|
|
40
|
+
api_key: str | None = None,
|
|
41
|
+
*,
|
|
42
|
+
model: str = "gpt-4o-mini",
|
|
43
|
+
base_url: str = "https://api.openai.com/v1",
|
|
44
|
+
request: RequestTransport | None = None,
|
|
45
|
+
timeout: float = 60.0,
|
|
46
|
+
) -> None:
|
|
47
|
+
self._api_key = api_key if api_key is not None else os.getenv("OPENAI_API_KEY")
|
|
48
|
+
self._model = model
|
|
49
|
+
self._base_url = base_url.rstrip("/")
|
|
50
|
+
self._request = request or default_request
|
|
51
|
+
self._timeout = timeout
|
|
52
|
+
|
|
53
|
+
def describe(self, image_path: str, prompt: str) -> str:
|
|
54
|
+
if not self._api_key:
|
|
55
|
+
raise ProviderError("vision requires an API key (set OPENAI_API_KEY)")
|
|
56
|
+
status, body = self._request(
|
|
57
|
+
"POST",
|
|
58
|
+
f"{self._base_url}/chat/completions",
|
|
59
|
+
json_body={
|
|
60
|
+
"model": self._model,
|
|
61
|
+
"messages": [
|
|
62
|
+
{
|
|
63
|
+
"role": "user",
|
|
64
|
+
"content": [
|
|
65
|
+
{"type": "text", "text": prompt},
|
|
66
|
+
{"type": "image_url", "image_url": {"url": _image_data_url(image_path)}},
|
|
67
|
+
],
|
|
68
|
+
}
|
|
69
|
+
],
|
|
70
|
+
},
|
|
71
|
+
headers={"Authorization": f"Bearer {self._api_key}"},
|
|
72
|
+
timeout=self._timeout,
|
|
73
|
+
)
|
|
74
|
+
_raise_for_status(status, body)
|
|
75
|
+
try:
|
|
76
|
+
return body["choices"][0]["message"]["content"]
|
|
77
|
+
except (KeyError, IndexError, TypeError) as exc:
|
|
78
|
+
raise ProviderError(f"vision returned an unexpected response: {body}") from exc
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class TesseractOCR:
|
|
82
|
+
"""Local OCR via Tesseract (lazy ``pytesseract`` + ``Pillow``)."""
|
|
83
|
+
|
|
84
|
+
name = "tesseract"
|
|
85
|
+
|
|
86
|
+
def __init__(self, lang: str = "eng") -> None:
|
|
87
|
+
self._lang = lang
|
|
88
|
+
|
|
89
|
+
def extract_text(self, image_path: str) -> str:
|
|
90
|
+
try:
|
|
91
|
+
pytesseract = importlib.import_module("pytesseract")
|
|
92
|
+
Image = importlib.import_module("PIL.Image")
|
|
93
|
+
except ImportError as exc:
|
|
94
|
+
raise ProviderError(
|
|
95
|
+
"OCR requires 'pytesseract' and 'pillow'; install with "
|
|
96
|
+
"'pip install xyberos-vision[ocr]' (plus the tesseract binary)"
|
|
97
|
+
) from exc
|
|
98
|
+
return pytesseract.image_to_string(Image.open(image_path), lang=self._lang).strip()
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class OpenAIImageGenerator:
|
|
102
|
+
"""Image generation via the OpenAI images API (b64_json)."""
|
|
103
|
+
|
|
104
|
+
name = "openai"
|
|
105
|
+
url = "https://api.openai.com/v1/images/generations"
|
|
106
|
+
|
|
107
|
+
def __init__(
|
|
108
|
+
self,
|
|
109
|
+
api_key: str | None = None,
|
|
110
|
+
*,
|
|
111
|
+
model: str = "dall-e-3",
|
|
112
|
+
size: str = "1024x1024",
|
|
113
|
+
request: RequestTransport | None = None,
|
|
114
|
+
timeout: float = 120.0,
|
|
115
|
+
) -> None:
|
|
116
|
+
self._api_key = api_key if api_key is not None else os.getenv("OPENAI_API_KEY")
|
|
117
|
+
self._model = model
|
|
118
|
+
self._size = size
|
|
119
|
+
self._request = request or default_request
|
|
120
|
+
self._timeout = timeout
|
|
121
|
+
|
|
122
|
+
def generate(self, prompt: str, output_path: str) -> str:
|
|
123
|
+
if not self._api_key:
|
|
124
|
+
raise ProviderError("image generation requires an API key (set OPENAI_API_KEY)")
|
|
125
|
+
status, body = self._request(
|
|
126
|
+
"POST",
|
|
127
|
+
self.url,
|
|
128
|
+
json_body={"model": self._model, "prompt": prompt, "n": 1, "size": self._size, "response_format": "b64_json"},
|
|
129
|
+
headers={"Authorization": f"Bearer {self._api_key}"},
|
|
130
|
+
timeout=self._timeout,
|
|
131
|
+
)
|
|
132
|
+
_raise_for_status(status, body)
|
|
133
|
+
try:
|
|
134
|
+
encoded = body["data"][0]["b64_json"]
|
|
135
|
+
except (KeyError, IndexError, TypeError) as exc:
|
|
136
|
+
raise ProviderError(f"image generation returned an unexpected response: {body}") from exc
|
|
137
|
+
path = Path(output_path)
|
|
138
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
139
|
+
path.write_bytes(base64.b64decode(encoded))
|
|
140
|
+
return str(path)
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Vision / OCR / image-generation contracts (RFC-0019, M8)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Protocol, runtime_checkable
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@runtime_checkable
|
|
9
|
+
class VisionModel(Protocol):
|
|
10
|
+
"""Anything that describes an image given a prompt."""
|
|
11
|
+
|
|
12
|
+
name: str
|
|
13
|
+
|
|
14
|
+
def describe(self, image_path: str, prompt: str) -> str:
|
|
15
|
+
"""Return a text description of the image at ``image_path``."""
|
|
16
|
+
...
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@runtime_checkable
|
|
20
|
+
class OCR(Protocol):
|
|
21
|
+
"""Anything that extracts text from an image."""
|
|
22
|
+
|
|
23
|
+
name: str
|
|
24
|
+
|
|
25
|
+
def extract_text(self, image_path: str) -> str:
|
|
26
|
+
"""Return the text found in the image at ``image_path``."""
|
|
27
|
+
...
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@runtime_checkable
|
|
31
|
+
class ImageGenerator(Protocol):
|
|
32
|
+
"""Anything that turns a prompt into an image file."""
|
|
33
|
+
|
|
34
|
+
name: str
|
|
35
|
+
|
|
36
|
+
def generate(self, prompt: str, output_path: str) -> str:
|
|
37
|
+
"""Write a generated image to ``output_path`` and return the path."""
|
|
38
|
+
...
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""A tiny stdlib HTTP helper (no third-party deps).
|
|
2
|
+
|
|
3
|
+
Supports JSON bodies, raw bytes bodies, and both parsed-JSON and raw-bytes
|
|
4
|
+
responses. Both transports are injectable so tests run without a network.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import urllib.error
|
|
11
|
+
import urllib.parse
|
|
12
|
+
import urllib.request
|
|
13
|
+
from typing import Any, Callable
|
|
14
|
+
|
|
15
|
+
#: (method, url, *, json_body, raw_body, headers, query, timeout) -> (status, body)
|
|
16
|
+
RequestTransport = Callable[..., tuple[int, Any]]
|
|
17
|
+
RawRequestTransport = Callable[..., tuple[int, bytes]]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _build(
|
|
21
|
+
method: str,
|
|
22
|
+
url: str,
|
|
23
|
+
*,
|
|
24
|
+
json_body: Any,
|
|
25
|
+
raw_body: bytes | None,
|
|
26
|
+
headers: dict[str, str] | None,
|
|
27
|
+
query: dict[str, Any] | None,
|
|
28
|
+
) -> urllib.request.Request:
|
|
29
|
+
final_url = url
|
|
30
|
+
if query:
|
|
31
|
+
separator = "&" if "?" in url else "?"
|
|
32
|
+
final_url = url + separator + urllib.parse.urlencode(query)
|
|
33
|
+
|
|
34
|
+
data: bytes | None = None
|
|
35
|
+
request_headers = dict(headers or {})
|
|
36
|
+
if raw_body is not None:
|
|
37
|
+
data = raw_body
|
|
38
|
+
elif json_body is not None:
|
|
39
|
+
data = json.dumps(json_body).encode("utf-8")
|
|
40
|
+
request_headers.setdefault("Content-Type", "application/json")
|
|
41
|
+
return urllib.request.Request(
|
|
42
|
+
final_url, data=data, headers=request_headers, method=method
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def default_request(
|
|
47
|
+
method: str,
|
|
48
|
+
url: str,
|
|
49
|
+
*,
|
|
50
|
+
json_body: Any = None,
|
|
51
|
+
raw_body: bytes | None = None,
|
|
52
|
+
headers: dict[str, str] | None = None,
|
|
53
|
+
query: dict[str, Any] | None = None,
|
|
54
|
+
timeout: float = 30.0,
|
|
55
|
+
) -> tuple[int, Any]:
|
|
56
|
+
"""Send one request and return ``(status, parsed_json_or_text)``."""
|
|
57
|
+
request = _build(method, url, json_body=json_body, raw_body=raw_body, headers=headers, query=query)
|
|
58
|
+
try:
|
|
59
|
+
with urllib.request.urlopen(request, timeout=timeout) as response:
|
|
60
|
+
raw = response.read()
|
|
61
|
+
content_type = response.headers.get("Content-Type", "")
|
|
62
|
+
except urllib.error.HTTPError as exc:
|
|
63
|
+
return exc.code, exc.read().decode("utf-8", errors="replace")
|
|
64
|
+
|
|
65
|
+
text = raw.decode("utf-8", errors="replace")
|
|
66
|
+
if "application/json" in content_type or text.lstrip().startswith("{"):
|
|
67
|
+
try:
|
|
68
|
+
return 200, json.loads(text)
|
|
69
|
+
except json.JSONDecodeError:
|
|
70
|
+
return 200, text
|
|
71
|
+
return 200, text
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def default_raw_request(
|
|
75
|
+
method: str,
|
|
76
|
+
url: str,
|
|
77
|
+
*,
|
|
78
|
+
json_body: Any = None,
|
|
79
|
+
raw_body: bytes | None = None,
|
|
80
|
+
headers: dict[str, str] | None = None,
|
|
81
|
+
query: dict[str, Any] | None = None,
|
|
82
|
+
timeout: float = 30.0,
|
|
83
|
+
) -> tuple[int, bytes]:
|
|
84
|
+
"""Send one request and return ``(status, raw_bytes)``."""
|
|
85
|
+
request = _build(method, url, json_body=json_body, raw_body=raw_body, headers=headers, query=query)
|
|
86
|
+
try:
|
|
87
|
+
with urllib.request.urlopen(request, timeout=timeout) as response:
|
|
88
|
+
return int(getattr(response, "status", 200)), response.read()
|
|
89
|
+
except urllib.error.HTTPError as exc:
|
|
90
|
+
return exc.code, exc.read()
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""Vision plugin entry point (RFC-0019, M8)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from typing import Any, cast
|
|
7
|
+
|
|
8
|
+
from xyberos.contracts import Plugin, Tool
|
|
9
|
+
from xyberos.tools import FunctionTool
|
|
10
|
+
|
|
11
|
+
from .adapters import OpenAICompatibleVision, OpenAIImageGenerator, TesseractOCR
|
|
12
|
+
from .http import RequestTransport
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _pop_tool(registry: Any, name: str) -> None:
|
|
16
|
+
unregister = getattr(registry, "unregister", None)
|
|
17
|
+
if callable(unregister):
|
|
18
|
+
unregister(name)
|
|
19
|
+
return
|
|
20
|
+
store = getattr(registry, "_tools", None)
|
|
21
|
+
if isinstance(store, dict):
|
|
22
|
+
cast(dict[str, Any], store).pop(name, None)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class VisionPlugin(Plugin):
|
|
26
|
+
"""Registers vision / OCR / image-generation tools."""
|
|
27
|
+
|
|
28
|
+
def __init__(
|
|
29
|
+
self,
|
|
30
|
+
api_key: str | None = None,
|
|
31
|
+
*,
|
|
32
|
+
env_prefix: str = "VISION",
|
|
33
|
+
request: RequestTransport | None = None,
|
|
34
|
+
) -> None:
|
|
35
|
+
self._api_key = api_key
|
|
36
|
+
self._env_prefix = env_prefix
|
|
37
|
+
self._request = request
|
|
38
|
+
self._model = os.getenv(f"{env_prefix}_MODEL", "gpt-4o-mini")
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def name(self) -> str:
|
|
42
|
+
return "vision"
|
|
43
|
+
|
|
44
|
+
def vision_model(self) -> OpenAICompatibleVision:
|
|
45
|
+
return OpenAICompatibleVision(self._api_key, model=self._model, request=self._request)
|
|
46
|
+
|
|
47
|
+
def ocr(self) -> TesseractOCR:
|
|
48
|
+
return TesseractOCR()
|
|
49
|
+
|
|
50
|
+
def image_generator(self) -> OpenAIImageGenerator:
|
|
51
|
+
return OpenAIImageGenerator(self._api_key, request=self._request)
|
|
52
|
+
|
|
53
|
+
def tools(self) -> list[Tool]:
|
|
54
|
+
vision = self.vision_model()
|
|
55
|
+
ocr = self.ocr()
|
|
56
|
+
generator = self.image_generator()
|
|
57
|
+
|
|
58
|
+
def _describe(image_path: str, prompt: str = "Describe this image in detail.") -> str:
|
|
59
|
+
return vision.describe(image_path, prompt)
|
|
60
|
+
|
|
61
|
+
def _ocr(image_path: str) -> str:
|
|
62
|
+
return ocr.extract_text(image_path)
|
|
63
|
+
|
|
64
|
+
def _generate(prompt: str, output_path: str) -> str:
|
|
65
|
+
return generator.generate(prompt, output_path)
|
|
66
|
+
|
|
67
|
+
return [
|
|
68
|
+
FunctionTool("vision_describe", _describe, description="Describe an image with a vision model."),
|
|
69
|
+
FunctionTool("ocr_extract_text", _ocr, description="Extract text from an image with OCR."),
|
|
70
|
+
FunctionTool("image_generate", _generate, description="Generate an image from a prompt."),
|
|
71
|
+
]
|
|
72
|
+
|
|
73
|
+
def register(self, kernel: object) -> None:
|
|
74
|
+
registry = kernel.resolve("tools")
|
|
75
|
+
for tool in self.tools():
|
|
76
|
+
registry.register(tool)
|
|
77
|
+
|
|
78
|
+
def unregister(self, kernel: object) -> None:
|
|
79
|
+
registry = kernel.resolve("tools")
|
|
80
|
+
for tool in self.tools():
|
|
81
|
+
_pop_tool(registry, tool.name)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
#: Auto-discovered by ``app.load_entry_points()``.
|
|
85
|
+
plugin = VisionPlugin()
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: xyberos-vision
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Vision plugin (RFC-0019, M8): describe images, OCR, generate images
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Keywords: xyberos,plugin,vision,ocr,tesseract,image
|
|
7
|
+
Requires-Python: >=3.10
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
Requires-Dist: xyberos>=1.0
|
|
10
|
+
Provides-Extra: ocr
|
|
11
|
+
Requires-Dist: pytesseract; extra == "ocr"
|
|
12
|
+
Requires-Dist: pillow; extra == "ocr"
|
|
13
|
+
Provides-Extra: test
|
|
14
|
+
Requires-Dist: pytest; extra == "test"
|
|
15
|
+
|
|
16
|
+
# xyberos-vision
|
|
17
|
+
|
|
18
|
+
**Vision plugin — RFC-0019, M8.** Describe images (vision model), extract text
|
|
19
|
+
(OCR), and generate images — three tools, three contracts.
|
|
20
|
+
|
|
21
|
+
## Install
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install -e ./vision
|
|
25
|
+
pip install xyberos-vision[ocr] # optional: pytesseract + pillow for OCR
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Usage
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from xyberos import create_app
|
|
32
|
+
from xyberos_vision import VisionPlugin
|
|
33
|
+
|
|
34
|
+
app = create_app()
|
|
35
|
+
app.load_plugin(VisionPlugin()) # key from OPENAI_API_KEY
|
|
36
|
+
|
|
37
|
+
app.tools.execute("vision_describe", None, image_path="photo.png", prompt="What's in this photo?")
|
|
38
|
+
app.tools.execute("ocr_extract_text", None, image_path="scan.png")
|
|
39
|
+
app.tools.execute("image_generate", None, prompt="a sunset over the ocean", output_path="out.png")
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Tools
|
|
43
|
+
|
|
44
|
+
| Tool | Backed by |
|
|
45
|
+
| ---- | --------- |
|
|
46
|
+
| `vision_describe(image_path, prompt=...)` | `OpenAICompatibleVision` (`VISION_MODEL`, default `gpt-4o-mini`) |
|
|
47
|
+
| `ocr_extract_text(image_path)` | `TesseractOCR` (lazy `pytesseract` + `pillow`) |
|
|
48
|
+
| `image_generate(prompt, output_path)` | `OpenAIImageGenerator` (`dall-e-3`) |
|
|
49
|
+
|
|
50
|
+
## Tests
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
pip install pytest
|
|
54
|
+
pytest tests/
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Vision/image-gen are tested via injectable transports; OCR tests skip when
|
|
58
|
+
`pytesseract` is absent.
|
|
59
|
+
|
|
60
|
+
## Ship location
|
|
61
|
+
|
|
62
|
+
Plugin (`xyberos.plugins` entry point) — vision/multimodal (M8).
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
tests/test_adapters.py
|
|
4
|
+
tests/test_plugin.py
|
|
5
|
+
xyberos_vision/__init__.py
|
|
6
|
+
xyberos_vision/adapters.py
|
|
7
|
+
xyberos_vision/contract.py
|
|
8
|
+
xyberos_vision/http.py
|
|
9
|
+
xyberos_vision/plugin.py
|
|
10
|
+
xyberos_vision.egg-info/PKG-INFO
|
|
11
|
+
xyberos_vision.egg-info/SOURCES.txt
|
|
12
|
+
xyberos_vision.egg-info/dependency_links.txt
|
|
13
|
+
xyberos_vision.egg-info/entry_points.txt
|
|
14
|
+
xyberos_vision.egg-info/requires.txt
|
|
15
|
+
xyberos_vision.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
xyberos_vision
|