xyberos-vision 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,62 @@
1
+ Metadata-Version: 2.4
2
+ Name: xyberos-vision
3
+ Version: 0.1.0
4
+ Summary: Vision plugin (RFC-0019, M8): describe images, OCR, generate images
5
+ License: Apache-2.0
6
+ Keywords: xyberos,plugin,vision,ocr,tesseract,image
7
+ Requires-Python: >=3.10
8
+ Description-Content-Type: text/markdown
9
+ Requires-Dist: xyberos>=1.0
10
+ Provides-Extra: ocr
11
+ Requires-Dist: pytesseract; extra == "ocr"
12
+ Requires-Dist: pillow; extra == "ocr"
13
+ Provides-Extra: test
14
+ Requires-Dist: pytest; extra == "test"
15
+
16
+ # xyberos-vision
17
+
18
+ **Vision plugin — RFC-0019, M8.** Describe images (vision model), extract text
19
+ (OCR), and generate images — three tools, three contracts.
20
+
21
+ ## Install
22
+
23
+ ```bash
24
+ pip install -e ./vision
25
+ pip install xyberos-vision[ocr] # optional: pytesseract + pillow for OCR
26
+ ```
27
+
28
+ ## Usage
29
+
30
+ ```python
31
+ from xyberos import create_app
32
+ from xyberos_vision import VisionPlugin
33
+
34
+ app = create_app()
35
+ app.load_plugin(VisionPlugin()) # key from OPENAI_API_KEY
36
+
37
+ app.tools.execute("vision_describe", None, image_path="photo.png", prompt="What's in this photo?")
38
+ app.tools.execute("ocr_extract_text", None, image_path="scan.png")
39
+ app.tools.execute("image_generate", None, prompt="a sunset over the ocean", output_path="out.png")
40
+ ```
41
+
42
+ ## Tools
43
+
44
+ | Tool | Backed by |
45
+ | ---- | --------- |
46
+ | `vision_describe(image_path, prompt=...)` | `OpenAICompatibleVision` (`VISION_MODEL`, default `gpt-4o-mini`) |
47
+ | `ocr_extract_text(image_path)` | `TesseractOCR` (lazy `pytesseract` + `pillow`) |
48
+ | `image_generate(prompt, output_path)` | `OpenAIImageGenerator` (`dall-e-3`) |
49
+
50
+ ## Tests
51
+
52
+ ```bash
53
+ pip install pytest
54
+ pytest tests/
55
+ ```
56
+
57
+ Vision/image-gen are tested via injectable transports; OCR tests skip when
58
+ `pytesseract` is absent.
59
+
60
+ ## Ship location
61
+
62
+ Plugin (`xyberos.plugins` entry point) — vision/multimodal (M8).
@@ -0,0 +1,47 @@
1
+ # xyberos-vision
2
+
3
+ **Vision plugin — RFC-0019, M8.** Describe images (vision model), extract text
4
+ (OCR), and generate images — three tools, three contracts.
5
+
6
+ ## Install
7
+
8
+ ```bash
9
+ pip install -e ./vision
10
+ pip install xyberos-vision[ocr] # optional: pytesseract + pillow for OCR
11
+ ```
12
+
13
+ ## Usage
14
+
15
+ ```python
16
+ from xyberos import create_app
17
+ from xyberos_vision import VisionPlugin
18
+
19
+ app = create_app()
20
+ app.load_plugin(VisionPlugin()) # key from OPENAI_API_KEY
21
+
22
+ app.tools.execute("vision_describe", None, image_path="photo.png", prompt="What's in this photo?")
23
+ app.tools.execute("ocr_extract_text", None, image_path="scan.png")
24
+ app.tools.execute("image_generate", None, prompt="a sunset over the ocean", output_path="out.png")
25
+ ```
26
+
27
+ ## Tools
28
+
29
+ | Tool | Backed by |
30
+ | ---- | --------- |
31
+ | `vision_describe(image_path, prompt=...)` | `OpenAICompatibleVision` (`VISION_MODEL`, default `gpt-4o-mini`) |
32
+ | `ocr_extract_text(image_path)` | `TesseractOCR` (lazy `pytesseract` + `pillow`) |
33
+ | `image_generate(prompt, output_path)` | `OpenAIImageGenerator` (`dall-e-3`) |
34
+
35
+ ## Tests
36
+
37
+ ```bash
38
+ pip install pytest
39
+ pytest tests/
40
+ ```
41
+
42
+ Vision/image-gen are tested via injectable transports; OCR tests skip when
43
+ `pytesseract` is absent.
44
+
45
+ ## Ship location
46
+
47
+ Plugin (`xyberos.plugins` entry point) — vision/multimodal (M8).
@@ -0,0 +1,27 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "xyberos-vision"
7
+ version = "0.1.0"
8
+ description = "Vision plugin (RFC-0019, M8): describe images, OCR, generate images"
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = {text = "Apache-2.0"}
12
+ dependencies = ["xyberos>=1.0"]
13
+ keywords = ["xyberos", "plugin", "vision", "ocr", "tesseract", "image"]
14
+
15
+ [project.optional-dependencies]
16
+ ocr = ["pytesseract", "pillow"]
17
+ test = ["pytest"]
18
+
19
+ [project.entry-points."xyberos.plugins"]
20
+ vision = "xyberos_vision.plugin:plugin"
21
+
22
+ [tool.setuptools]
23
+ packages = ["xyberos_vision"]
24
+
25
+ [tool.pytest.ini_options]
26
+ testpaths = ["tests"]
27
+ pythonpath = ["."]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,56 @@
1
+ """Tests for the vision adapters (injectable transports / skips, no network)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import base64
6
+ import importlib.util
7
+
8
+ import pytest
9
+ from xyberos.exceptions.provider import ProviderError
10
+
11
+ from xyberos_vision import OpenAICompatibleVision, OpenAIImageGenerator, TesseractOCR
12
+
13
+
14
+ def _tiny_png(tmp_path) -> str:
15
+ # A 1x1 PNG so image reading + base64 works without real image deps.
16
+ path = tmp_path / "img.png"
17
+ path.write_bytes(b"\x89PNG\r\n\x1a\n\x00\x00\x00\rioEND")
18
+ return str(path)
19
+
20
+
21
+ def test_vision_describe(tmp_path):
22
+ captured = {}
23
+
24
+ def request(method, url, **kwargs):
25
+ captured.update(kwargs)
26
+ return 200, {"choices": [{"message": {"content": "A cat"}}]}
27
+
28
+ image = _tiny_png(tmp_path)
29
+ vision = OpenAICompatibleVision(api_key="k", request=request)
30
+ assert vision.describe(image, "What is this?") == "A cat"
31
+ content = captured["json_body"]["messages"][0]["content"]
32
+ assert content[0]["text"] == "What is this?"
33
+ assert content[1]["type"] == "image_url"
34
+ assert content[1]["image_url"]["url"].startswith("data:image/png;base64,")
35
+
36
+
37
+ def test_vision_requires_key():
38
+ with pytest.raises(ProviderError, match="OPENAI_API_KEY"):
39
+ OpenAICompatibleVision(api_key=None, request=lambda *a, **k: (200, {})).describe("x.png", "p")
40
+
41
+
42
+ def test_image_generate(tmp_path):
43
+ def request(method, url, **kwargs):
44
+ return 200, {"data": [{"b64_json": base64.b64encode(b"PNGDATA").decode("ascii")}]}
45
+
46
+ output = tmp_path / "gen.png"
47
+ gen = OpenAIImageGenerator(api_key="k", request=request)
48
+ assert gen.generate("a dog", str(output)) == str(output)
49
+ assert output.read_bytes() == b"PNGDATA"
50
+
51
+
52
+ def test_ocr():
53
+ if importlib.util.find_spec("pytesseract") is None:
54
+ pytest.skip("pytesseract is not installed")
55
+ ocr = TesseractOCR()
56
+ assert isinstance(ocr, TesseractOCR)
@@ -0,0 +1,32 @@
1
+ """Tests for loading the vision plugin into a Xyberos app."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import base64
6
+
7
+ from xyberos import create_app
8
+
9
+ from xyberos_vision import VisionPlugin
10
+
11
+
12
+ def test_plugin_registers_tools(tmp_path):
13
+ def request(method, url, **kwargs):
14
+ if url.endswith("/images/generations"):
15
+ return 200, {"data": [{"b64_json": base64.b64encode(b"PNG").decode("ascii")}]}
16
+ return 200, {"choices": [{"message": {"content": "A cat"}}]}
17
+
18
+ image = tmp_path / "img.png"
19
+ image.write_bytes(b"\x89PNG\r\n\x1a\n\x00\x00\x00\rioEND")
20
+
21
+ app = create_app()
22
+ app.load_plugin(VisionPlugin(api_key="k", request=request))
23
+ assert "vision_describe" in app.tools.names
24
+ assert "ocr_extract_text" in app.tools.names
25
+ assert "image_generate" in app.tools.names
26
+
27
+ assert app.tools.execute("vision_describe", None, image_path=str(image)) == "A cat"
28
+ output = tmp_path / "gen.png"
29
+ assert app.tools.execute("image_generate", None, prompt="a dog", output_path=str(output)) == str(output)
30
+ assert output.read_bytes() == b"PNG"
31
+
32
+ app.unload_plugin("vision")
@@ -0,0 +1,15 @@
1
+ """Vision plugin (RFC-0019, M8)."""
2
+
3
+ from .adapters import OpenAICompatibleVision, OpenAIImageGenerator, TesseractOCR
4
+ from .contract import ImageGenerator, OCR, VisionModel
5
+ from .plugin import VisionPlugin
6
+
7
+ __all__ = [
8
+ "ImageGenerator",
9
+ "OCR",
10
+ "OpenAICompatibleVision",
11
+ "OpenAIImageGenerator",
12
+ "TesseractOCR",
13
+ "VisionModel",
14
+ "VisionPlugin",
15
+ ]
@@ -0,0 +1,140 @@
1
+ """Vision adapters: OpenAI-compatible vision, Tesseract OCR, image generation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import base64
6
+ import importlib
7
+ import mimetypes
8
+ import os
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+ from xyberos.exceptions.provider import ProviderError
13
+
14
+ from .http import RequestTransport, default_request
15
+
16
+
17
+ def _raise_for_status(status: int, body: Any) -> None:
18
+ if 200 <= status < 300:
19
+ return
20
+ message = body if isinstance(body, str) else str(body)
21
+ raise ProviderError(f"vision API returned HTTP {status}: {message[:200]}")
22
+
23
+
24
+ def _image_data_url(image_path: str) -> str:
25
+ path = Path(image_path)
26
+ if not path.is_file():
27
+ raise FileNotFoundError(image_path)
28
+ mime = mimetypes.guess_type(str(path))[0] or "image/png"
29
+ encoded = base64.b64encode(path.read_bytes()).decode("ascii")
30
+ return f"data:{mime};base64,{encoded}"
31
+
32
+
33
+ class OpenAICompatibleVision:
34
+ """Chat-completions vision model (works with OpenAI / Gemini-compatible)."""
35
+
36
+ name = "openai_compatible"
37
+
38
+ def __init__(
39
+ self,
40
+ api_key: str | None = None,
41
+ *,
42
+ model: str = "gpt-4o-mini",
43
+ base_url: str = "https://api.openai.com/v1",
44
+ request: RequestTransport | None = None,
45
+ timeout: float = 60.0,
46
+ ) -> None:
47
+ self._api_key = api_key if api_key is not None else os.getenv("OPENAI_API_KEY")
48
+ self._model = model
49
+ self._base_url = base_url.rstrip("/")
50
+ self._request = request or default_request
51
+ self._timeout = timeout
52
+
53
+ def describe(self, image_path: str, prompt: str) -> str:
54
+ if not self._api_key:
55
+ raise ProviderError("vision requires an API key (set OPENAI_API_KEY)")
56
+ status, body = self._request(
57
+ "POST",
58
+ f"{self._base_url}/chat/completions",
59
+ json_body={
60
+ "model": self._model,
61
+ "messages": [
62
+ {
63
+ "role": "user",
64
+ "content": [
65
+ {"type": "text", "text": prompt},
66
+ {"type": "image_url", "image_url": {"url": _image_data_url(image_path)}},
67
+ ],
68
+ }
69
+ ],
70
+ },
71
+ headers={"Authorization": f"Bearer {self._api_key}"},
72
+ timeout=self._timeout,
73
+ )
74
+ _raise_for_status(status, body)
75
+ try:
76
+ return body["choices"][0]["message"]["content"]
77
+ except (KeyError, IndexError, TypeError) as exc:
78
+ raise ProviderError(f"vision returned an unexpected response: {body}") from exc
79
+
80
+
81
+ class TesseractOCR:
82
+ """Local OCR via Tesseract (lazy ``pytesseract`` + ``Pillow``)."""
83
+
84
+ name = "tesseract"
85
+
86
+ def __init__(self, lang: str = "eng") -> None:
87
+ self._lang = lang
88
+
89
+ def extract_text(self, image_path: str) -> str:
90
+ try:
91
+ pytesseract = importlib.import_module("pytesseract")
92
+ Image = importlib.import_module("PIL.Image")
93
+ except ImportError as exc:
94
+ raise ProviderError(
95
+ "OCR requires 'pytesseract' and 'pillow'; install with "
96
+ "'pip install xyberos-vision[ocr]' (plus the tesseract binary)"
97
+ ) from exc
98
+ return pytesseract.image_to_string(Image.open(image_path), lang=self._lang).strip()
99
+
100
+
101
+ class OpenAIImageGenerator:
102
+ """Image generation via the OpenAI images API (b64_json)."""
103
+
104
+ name = "openai"
105
+ url = "https://api.openai.com/v1/images/generations"
106
+
107
+ def __init__(
108
+ self,
109
+ api_key: str | None = None,
110
+ *,
111
+ model: str = "dall-e-3",
112
+ size: str = "1024x1024",
113
+ request: RequestTransport | None = None,
114
+ timeout: float = 120.0,
115
+ ) -> None:
116
+ self._api_key = api_key if api_key is not None else os.getenv("OPENAI_API_KEY")
117
+ self._model = model
118
+ self._size = size
119
+ self._request = request or default_request
120
+ self._timeout = timeout
121
+
122
+ def generate(self, prompt: str, output_path: str) -> str:
123
+ if not self._api_key:
124
+ raise ProviderError("image generation requires an API key (set OPENAI_API_KEY)")
125
+ status, body = self._request(
126
+ "POST",
127
+ self.url,
128
+ json_body={"model": self._model, "prompt": prompt, "n": 1, "size": self._size, "response_format": "b64_json"},
129
+ headers={"Authorization": f"Bearer {self._api_key}"},
130
+ timeout=self._timeout,
131
+ )
132
+ _raise_for_status(status, body)
133
+ try:
134
+ encoded = body["data"][0]["b64_json"]
135
+ except (KeyError, IndexError, TypeError) as exc:
136
+ raise ProviderError(f"image generation returned an unexpected response: {body}") from exc
137
+ path = Path(output_path)
138
+ path.parent.mkdir(parents=True, exist_ok=True)
139
+ path.write_bytes(base64.b64decode(encoded))
140
+ return str(path)
@@ -0,0 +1,38 @@
1
+ """Vision / OCR / image-generation contracts (RFC-0019, M8)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Protocol, runtime_checkable
6
+
7
+
8
+ @runtime_checkable
9
+ class VisionModel(Protocol):
10
+ """Anything that describes an image given a prompt."""
11
+
12
+ name: str
13
+
14
+ def describe(self, image_path: str, prompt: str) -> str:
15
+ """Return a text description of the image at ``image_path``."""
16
+ ...
17
+
18
+
19
+ @runtime_checkable
20
+ class OCR(Protocol):
21
+ """Anything that extracts text from an image."""
22
+
23
+ name: str
24
+
25
+ def extract_text(self, image_path: str) -> str:
26
+ """Return the text found in the image at ``image_path``."""
27
+ ...
28
+
29
+
30
+ @runtime_checkable
31
+ class ImageGenerator(Protocol):
32
+ """Anything that turns a prompt into an image file."""
33
+
34
+ name: str
35
+
36
+ def generate(self, prompt: str, output_path: str) -> str:
37
+ """Write a generated image to ``output_path`` and return the path."""
38
+ ...
@@ -0,0 +1,90 @@
1
+ """A tiny stdlib HTTP helper (no third-party deps).
2
+
3
+ Supports JSON bodies, raw bytes bodies, and both parsed-JSON and raw-bytes
4
+ responses. Both transports are injectable so tests run without a network.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import urllib.error
11
+ import urllib.parse
12
+ import urllib.request
13
+ from typing import Any, Callable
14
+
15
+ #: (method, url, *, json_body, raw_body, headers, query, timeout) -> (status, body)
16
+ RequestTransport = Callable[..., tuple[int, Any]]
17
+ RawRequestTransport = Callable[..., tuple[int, bytes]]
18
+
19
+
20
+ def _build(
21
+ method: str,
22
+ url: str,
23
+ *,
24
+ json_body: Any,
25
+ raw_body: bytes | None,
26
+ headers: dict[str, str] | None,
27
+ query: dict[str, Any] | None,
28
+ ) -> urllib.request.Request:
29
+ final_url = url
30
+ if query:
31
+ separator = "&" if "?" in url else "?"
32
+ final_url = url + separator + urllib.parse.urlencode(query)
33
+
34
+ data: bytes | None = None
35
+ request_headers = dict(headers or {})
36
+ if raw_body is not None:
37
+ data = raw_body
38
+ elif json_body is not None:
39
+ data = json.dumps(json_body).encode("utf-8")
40
+ request_headers.setdefault("Content-Type", "application/json")
41
+ return urllib.request.Request(
42
+ final_url, data=data, headers=request_headers, method=method
43
+ )
44
+
45
+
46
+ def default_request(
47
+ method: str,
48
+ url: str,
49
+ *,
50
+ json_body: Any = None,
51
+ raw_body: bytes | None = None,
52
+ headers: dict[str, str] | None = None,
53
+ query: dict[str, Any] | None = None,
54
+ timeout: float = 30.0,
55
+ ) -> tuple[int, Any]:
56
+ """Send one request and return ``(status, parsed_json_or_text)``."""
57
+ request = _build(method, url, json_body=json_body, raw_body=raw_body, headers=headers, query=query)
58
+ try:
59
+ with urllib.request.urlopen(request, timeout=timeout) as response:
60
+ raw = response.read()
61
+ content_type = response.headers.get("Content-Type", "")
62
+ except urllib.error.HTTPError as exc:
63
+ return exc.code, exc.read().decode("utf-8", errors="replace")
64
+
65
+ text = raw.decode("utf-8", errors="replace")
66
+ if "application/json" in content_type or text.lstrip().startswith("{"):
67
+ try:
68
+ return 200, json.loads(text)
69
+ except json.JSONDecodeError:
70
+ return 200, text
71
+ return 200, text
72
+
73
+
74
+ def default_raw_request(
75
+ method: str,
76
+ url: str,
77
+ *,
78
+ json_body: Any = None,
79
+ raw_body: bytes | None = None,
80
+ headers: dict[str, str] | None = None,
81
+ query: dict[str, Any] | None = None,
82
+ timeout: float = 30.0,
83
+ ) -> tuple[int, bytes]:
84
+ """Send one request and return ``(status, raw_bytes)``."""
85
+ request = _build(method, url, json_body=json_body, raw_body=raw_body, headers=headers, query=query)
86
+ try:
87
+ with urllib.request.urlopen(request, timeout=timeout) as response:
88
+ return int(getattr(response, "status", 200)), response.read()
89
+ except urllib.error.HTTPError as exc:
90
+ return exc.code, exc.read()
@@ -0,0 +1,85 @@
1
+ """Vision plugin entry point (RFC-0019, M8)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ from typing import Any, cast
7
+
8
+ from xyberos.contracts import Plugin, Tool
9
+ from xyberos.tools import FunctionTool
10
+
11
+ from .adapters import OpenAICompatibleVision, OpenAIImageGenerator, TesseractOCR
12
+ from .http import RequestTransport
13
+
14
+
15
+ def _pop_tool(registry: Any, name: str) -> None:
16
+ unregister = getattr(registry, "unregister", None)
17
+ if callable(unregister):
18
+ unregister(name)
19
+ return
20
+ store = getattr(registry, "_tools", None)
21
+ if isinstance(store, dict):
22
+ cast(dict[str, Any], store).pop(name, None)
23
+
24
+
25
+ class VisionPlugin(Plugin):
26
+ """Registers vision / OCR / image-generation tools."""
27
+
28
+ def __init__(
29
+ self,
30
+ api_key: str | None = None,
31
+ *,
32
+ env_prefix: str = "VISION",
33
+ request: RequestTransport | None = None,
34
+ ) -> None:
35
+ self._api_key = api_key
36
+ self._env_prefix = env_prefix
37
+ self._request = request
38
+ self._model = os.getenv(f"{env_prefix}_MODEL", "gpt-4o-mini")
39
+
40
+ @property
41
+ def name(self) -> str:
42
+ return "vision"
43
+
44
+ def vision_model(self) -> OpenAICompatibleVision:
45
+ return OpenAICompatibleVision(self._api_key, model=self._model, request=self._request)
46
+
47
+ def ocr(self) -> TesseractOCR:
48
+ return TesseractOCR()
49
+
50
+ def image_generator(self) -> OpenAIImageGenerator:
51
+ return OpenAIImageGenerator(self._api_key, request=self._request)
52
+
53
+ def tools(self) -> list[Tool]:
54
+ vision = self.vision_model()
55
+ ocr = self.ocr()
56
+ generator = self.image_generator()
57
+
58
+ def _describe(image_path: str, prompt: str = "Describe this image in detail.") -> str:
59
+ return vision.describe(image_path, prompt)
60
+
61
+ def _ocr(image_path: str) -> str:
62
+ return ocr.extract_text(image_path)
63
+
64
+ def _generate(prompt: str, output_path: str) -> str:
65
+ return generator.generate(prompt, output_path)
66
+
67
+ return [
68
+ FunctionTool("vision_describe", _describe, description="Describe an image with a vision model."),
69
+ FunctionTool("ocr_extract_text", _ocr, description="Extract text from an image with OCR."),
70
+ FunctionTool("image_generate", _generate, description="Generate an image from a prompt."),
71
+ ]
72
+
73
+ def register(self, kernel: object) -> None:
74
+ registry = kernel.resolve("tools")
75
+ for tool in self.tools():
76
+ registry.register(tool)
77
+
78
+ def unregister(self, kernel: object) -> None:
79
+ registry = kernel.resolve("tools")
80
+ for tool in self.tools():
81
+ _pop_tool(registry, tool.name)
82
+
83
+
84
+ #: Auto-discovered by ``app.load_entry_points()``.
85
+ plugin = VisionPlugin()
@@ -0,0 +1,62 @@
1
+ Metadata-Version: 2.4
2
+ Name: xyberos-vision
3
+ Version: 0.1.0
4
+ Summary: Vision plugin (RFC-0019, M8): describe images, OCR, generate images
5
+ License: Apache-2.0
6
+ Keywords: xyberos,plugin,vision,ocr,tesseract,image
7
+ Requires-Python: >=3.10
8
+ Description-Content-Type: text/markdown
9
+ Requires-Dist: xyberos>=1.0
10
+ Provides-Extra: ocr
11
+ Requires-Dist: pytesseract; extra == "ocr"
12
+ Requires-Dist: pillow; extra == "ocr"
13
+ Provides-Extra: test
14
+ Requires-Dist: pytest; extra == "test"
15
+
16
+ # xyberos-vision
17
+
18
+ **Vision plugin — RFC-0019, M8.** Describe images (vision model), extract text
19
+ (OCR), and generate images — three tools, three contracts.
20
+
21
+ ## Install
22
+
23
+ ```bash
24
+ pip install -e ./vision
25
+ pip install xyberos-vision[ocr] # optional: pytesseract + pillow for OCR
26
+ ```
27
+
28
+ ## Usage
29
+
30
+ ```python
31
+ from xyberos import create_app
32
+ from xyberos_vision import VisionPlugin
33
+
34
+ app = create_app()
35
+ app.load_plugin(VisionPlugin()) # key from OPENAI_API_KEY
36
+
37
+ app.tools.execute("vision_describe", None, image_path="photo.png", prompt="What's in this photo?")
38
+ app.tools.execute("ocr_extract_text", None, image_path="scan.png")
39
+ app.tools.execute("image_generate", None, prompt="a sunset over the ocean", output_path="out.png")
40
+ ```
41
+
42
+ ## Tools
43
+
44
+ | Tool | Backed by |
45
+ | ---- | --------- |
46
+ | `vision_describe(image_path, prompt=...)` | `OpenAICompatibleVision` (`VISION_MODEL`, default `gpt-4o-mini`) |
47
+ | `ocr_extract_text(image_path)` | `TesseractOCR` (lazy `pytesseract` + `pillow`) |
48
+ | `image_generate(prompt, output_path)` | `OpenAIImageGenerator` (`dall-e-3`) |
49
+
50
+ ## Tests
51
+
52
+ ```bash
53
+ pip install pytest
54
+ pytest tests/
55
+ ```
56
+
57
+ Vision/image-gen are tested via injectable transports; OCR tests skip when
58
+ `pytesseract` is absent.
59
+
60
+ ## Ship location
61
+
62
+ Plugin (`xyberos.plugins` entry point) — vision/multimodal (M8).
@@ -0,0 +1,15 @@
1
+ README.md
2
+ pyproject.toml
3
+ tests/test_adapters.py
4
+ tests/test_plugin.py
5
+ xyberos_vision/__init__.py
6
+ xyberos_vision/adapters.py
7
+ xyberos_vision/contract.py
8
+ xyberos_vision/http.py
9
+ xyberos_vision/plugin.py
10
+ xyberos_vision.egg-info/PKG-INFO
11
+ xyberos_vision.egg-info/SOURCES.txt
12
+ xyberos_vision.egg-info/dependency_links.txt
13
+ xyberos_vision.egg-info/entry_points.txt
14
+ xyberos_vision.egg-info/requires.txt
15
+ xyberos_vision.egg-info/top_level.txt
@@ -0,0 +1,2 @@
1
+ [xyberos.plugins]
2
+ vision = xyberos_vision.plugin:plugin
@@ -0,0 +1,8 @@
1
+ xyberos>=1.0
2
+
3
+ [ocr]
4
+ pytesseract
5
+ pillow
6
+
7
+ [test]
8
+ pytest
@@ -0,0 +1 @@
1
+ xyberos_vision