spot-sdk-python 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spot_sdk/__init__.py +86 -0
- spot_sdk/analysis_context.py +99 -0
- spot_sdk/analyzer.py +106 -0
- spot_sdk/analyzer_base.py +316 -0
- spot_sdk/api_gateway.py +271 -0
- spot_sdk/config.py +46 -0
- spot_sdk/config_client.py +203 -0
- spot_sdk/config_helpers.py +25 -0
- spot_sdk/email.py +136 -0
- spot_sdk/errors.py +24 -0
- spot_sdk/knowledge.py +341 -0
- spot_sdk/knowledge_tags.py +31 -0
- spot_sdk/logging.py +133 -0
- spot_sdk/ollama.py +58 -0
- spot_sdk/orchestrator.py +70 -0
- spot_sdk/plugin.py +30 -0
- spot_sdk/results.py +129 -0
- spot_sdk/settings_schema.py +56 -0
- spot_sdk/testing/README.md +83 -0
- spot_sdk/testing/__init__.py +22 -0
- spot_sdk/testing/factories.py +177 -0
- spot_sdk/testing/fake_knowledge_client.py +105 -0
- spot_sdk/threat_levels.py +33 -0
- spot_sdk/workflow.py +139 -0
- spot_sdk_python-1.0.0.dist-info/METADATA +353 -0
- spot_sdk_python-1.0.0.dist-info/RECORD +27 -0
- spot_sdk_python-1.0.0.dist-info/WHEEL +4 -0
spot_sdk/knowledge.py
ADDED
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
"""Knowledge Store SDK: document model, client, chunk helper.
|
|
2
|
+
|
|
3
|
+
The Knowledge Store is a SPOT platform service where context providers
|
|
4
|
+
deposit tagged company documents (employees, wikis, policies, ...) and
|
|
5
|
+
analyzers fetch them on demand by semantic similarity + tag expression.
|
|
6
|
+
|
|
7
|
+
Neither side knows the other exists. Plugins only see this SDK; the
|
|
8
|
+
store owns embedding and vector search.
|
|
9
|
+
|
|
10
|
+
Quick start (ingestion provider)::
|
|
11
|
+
|
|
12
|
+
from spot_sdk.knowledge import KnowledgeClient, KnowledgeDocument
|
|
13
|
+
from spot_sdk.knowledge_tags import KnowledgeTag
|
|
14
|
+
|
|
15
|
+
kb = KnowledgeClient(
|
|
16
|
+
url=os.environ["SPOT_KNOWLEDGE_URL"],
|
|
17
|
+
api_key=os.environ.get("SPOT_INTERNAL_API_KEY"),
|
|
18
|
+
)
|
|
19
|
+
await kb.bulk_upsert([
|
|
20
|
+
KnowledgeDocument(
|
|
21
|
+
id=f"employee:{e.email}",
|
|
22
|
+
content=f"{e.name}, {e.title}, {e.department}",
|
|
23
|
+
tags=[KnowledgeTag.EMPLOYEE, KnowledgeTag.EXECUTIVE],
|
|
24
|
+
metadata={"email": e.email, "title": e.title},
|
|
25
|
+
source="provider-employee-dir",
|
|
26
|
+
)
|
|
27
|
+
for e in employees
|
|
28
|
+
])
|
|
29
|
+
|
|
30
|
+
Quick start (analyzer)::
|
|
31
|
+
|
|
32
|
+
from spot_sdk.knowledge import KnowledgeClient
|
|
33
|
+
|
|
34
|
+
@app.post("/internal/analyze")
|
|
35
|
+
async def analyze(email: Email) -> AnalysisResult:
|
|
36
|
+
kb = KnowledgeClient.for_analysis(email)
|
|
37
|
+
execs = await kb.fetch(tags="employee+executive", top_k=3)
|
|
38
|
+
...
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
from __future__ import annotations
|
|
42
|
+
|
|
43
|
+
import hashlib
|
|
44
|
+
import os
|
|
45
|
+
from datetime import datetime
|
|
46
|
+
from typing import TYPE_CHECKING, Any, cast
|
|
47
|
+
|
|
48
|
+
import httpx
|
|
49
|
+
from pydantic import BaseModel, Field
|
|
50
|
+
|
|
51
|
+
from .logging import get_logger
|
|
52
|
+
|
|
53
|
+
if TYPE_CHECKING:
|
|
54
|
+
from .email import Email
|
|
55
|
+
|
|
56
|
+
logger = get_logger(__name__)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# ----------------------------------------------------------------------- #
|
|
60
|
+
# Models
|
|
61
|
+
# ----------------------------------------------------------------------- #
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class KnowledgeDocument(BaseModel):
|
|
65
|
+
"""A document stored in (or returned from) the SPOT Knowledge Store.
|
|
66
|
+
|
|
67
|
+
There is no separate ``type`` field — ``tags`` is the single
|
|
68
|
+
categorisation axis. Use the constants on
|
|
69
|
+
:class:`spot_sdk.knowledge_tags.KnowledgeTag` for interoperable
|
|
70
|
+
common labels.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
id: str = Field(
|
|
74
|
+
...,
|
|
75
|
+
description=(
|
|
76
|
+
"Stable unique id (e.g. 'employee:alice@co.com', "
|
|
77
|
+
"'wiki:page-123#chunk-4'). Upsert is idempotent on this key."
|
|
78
|
+
),
|
|
79
|
+
)
|
|
80
|
+
content: str = Field(
|
|
81
|
+
...,
|
|
82
|
+
description=(
|
|
83
|
+
"Text payload an LLM or analyzer reads. Should be chunk-sized "
|
|
84
|
+
"for long documents; use spot_sdk.knowledge.chunk_text() to split."
|
|
85
|
+
),
|
|
86
|
+
)
|
|
87
|
+
tags: list[str] = Field(
|
|
88
|
+
default_factory=list,
|
|
89
|
+
description="Tag set; AND/OR-filterable at query time via tag expressions.",
|
|
90
|
+
)
|
|
91
|
+
metadata: dict[str, Any] = Field(
|
|
92
|
+
default_factory=dict,
|
|
93
|
+
description=(
|
|
94
|
+
"Structured fields (role, email, url, ...). "
|
|
95
|
+
"Reserved key 'parent_id' links chunks back to their source doc."
|
|
96
|
+
),
|
|
97
|
+
)
|
|
98
|
+
source: str = Field(
|
|
99
|
+
default="",
|
|
100
|
+
description="Provider that produced the document (audit + future ownership).",
|
|
101
|
+
)
|
|
102
|
+
updated_at: datetime | None = None
|
|
103
|
+
expires_at: datetime | None = Field(
|
|
104
|
+
default=None,
|
|
105
|
+
description="Optional TTL; the cleanup job deletes rows where expires_at < now().",
|
|
106
|
+
)
|
|
107
|
+
score: float | None = Field(
|
|
108
|
+
default=None,
|
|
109
|
+
ge=0.0,
|
|
110
|
+
le=1.0,
|
|
111
|
+
description="Populated on query results, 0.0 – 1.0 (cosine similarity).",
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
# ----------------------------------------------------------------------- #
|
|
116
|
+
# Chunking helper
|
|
117
|
+
# ----------------------------------------------------------------------- #
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def chunk_text(
|
|
121
|
+
text: str,
|
|
122
|
+
max_chars: int = 2000,
|
|
123
|
+
overlap: int = 200,
|
|
124
|
+
) -> list[str]:
|
|
125
|
+
"""Split a long text into overlapping character-bounded chunks.
|
|
126
|
+
|
|
127
|
+
Naive but predictable: walk the string in steps of (max_chars - overlap),
|
|
128
|
+
cut at the next whitespace within ``max_chars`` to avoid mid-word splits.
|
|
129
|
+
|
|
130
|
+
Args:
|
|
131
|
+
text: The text to split.
|
|
132
|
+
max_chars: Maximum chunk size in characters.
|
|
133
|
+
overlap: Characters of overlap between consecutive chunks.
|
|
134
|
+
|
|
135
|
+
Returns:
|
|
136
|
+
List of chunk strings. Empty input returns ``[]``.
|
|
137
|
+
"""
|
|
138
|
+
if not text:
|
|
139
|
+
return []
|
|
140
|
+
if max_chars <= 0:
|
|
141
|
+
raise ValueError("max_chars must be positive")
|
|
142
|
+
if overlap < 0 or overlap >= max_chars:
|
|
143
|
+
raise ValueError("overlap must be in [0, max_chars)")
|
|
144
|
+
|
|
145
|
+
text = text.strip()
|
|
146
|
+
if len(text) <= max_chars:
|
|
147
|
+
return [text]
|
|
148
|
+
|
|
149
|
+
chunks: list[str] = []
|
|
150
|
+
step = max_chars - overlap
|
|
151
|
+
start = 0
|
|
152
|
+
n = len(text)
|
|
153
|
+
|
|
154
|
+
while start < n:
|
|
155
|
+
end = min(start + max_chars, n)
|
|
156
|
+
# Try to cut on whitespace if we're not already at the end
|
|
157
|
+
if end < n:
|
|
158
|
+
cut = text.rfind(" ", start, end)
|
|
159
|
+
if cut > start + (max_chars // 2):
|
|
160
|
+
end = cut
|
|
161
|
+
chunks.append(text[start:end].strip())
|
|
162
|
+
if end >= n:
|
|
163
|
+
break
|
|
164
|
+
start = max(start + step, end - overlap)
|
|
165
|
+
|
|
166
|
+
return [c for c in chunks if c]
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
# ----------------------------------------------------------------------- #
|
|
170
|
+
# Client
|
|
171
|
+
# ----------------------------------------------------------------------- #
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
class KnowledgeClient:
|
|
175
|
+
"""HTTP client for the SPOT Knowledge Store.
|
|
176
|
+
|
|
177
|
+
Same class for both write side (providers: ``upsert``,
|
|
178
|
+
``bulk_upsert``, ``delete``) and read side (analyzers: ``fetch``).
|
|
179
|
+
All HTTP / embedding / vector / DB mechanics happen on the server;
|
|
180
|
+
callers only see typed Python objects.
|
|
181
|
+
|
|
182
|
+
Use :meth:`for_analysis` from analyzer code — it builds a client
|
|
183
|
+
pre-configured with the URL from env and the workflow-imposed
|
|
184
|
+
retrieval limits passed in by the orchestrator.
|
|
185
|
+
"""
|
|
186
|
+
|
|
187
|
+
DEFAULT_TIMEOUT = 15.0
|
|
188
|
+
|
|
189
|
+
def __init__(
|
|
190
|
+
self,
|
|
191
|
+
url: str,
|
|
192
|
+
api_key: str | None = None,
|
|
193
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
194
|
+
retrieval_limits: dict[str, Any] | None = None,
|
|
195
|
+
) -> None:
|
|
196
|
+
"""Initialize the client.
|
|
197
|
+
|
|
198
|
+
Args:
|
|
199
|
+
url: Base URL of the Knowledge Store API
|
|
200
|
+
(e.g. ``http://knowledge:8000/api/v1/knowledge``).
|
|
201
|
+
api_key: Optional ``SPOT_INTERNAL_API_KEY`` for write endpoints.
|
|
202
|
+
timeout: Request timeout in seconds.
|
|
203
|
+
retrieval_limits: Optional caps applied to ``fetch()`` —
|
|
204
|
+
``{"max_top_k": int, "min_score_floor": float}``. Only
|
|
205
|
+
shrinks what the caller asked for; never expands.
|
|
206
|
+
"""
|
|
207
|
+
self.url = url.rstrip("/")
|
|
208
|
+
self.api_key = api_key
|
|
209
|
+
self.timeout = timeout
|
|
210
|
+
self._limits = dict(retrieval_limits or {})
|
|
211
|
+
|
|
212
|
+
# ------------------------------------------------------------------ #
|
|
213
|
+
# Factories
|
|
214
|
+
# ------------------------------------------------------------------ #
|
|
215
|
+
|
|
216
|
+
@classmethod
|
|
217
|
+
def for_analysis(cls, email: "Email") -> KnowledgeClient:
|
|
218
|
+
"""Build a client pre-configured for the current analysis call.
|
|
219
|
+
|
|
220
|
+
Reads the URL from the ``SPOT_KNOWLEDGE_URL`` env var (injected
|
|
221
|
+
by the SPOT installer at plugin install time) and any retrieval
|
|
222
|
+
caps from ``email.retrieval_limits`` (set by the orchestrator
|
|
223
|
+
from the workflow stage's ``retrieval_limits``).
|
|
224
|
+
"""
|
|
225
|
+
url = os.environ.get("SPOT_KNOWLEDGE_URL")
|
|
226
|
+
if not url:
|
|
227
|
+
raise RuntimeError(
|
|
228
|
+
"SPOT_KNOWLEDGE_URL is not set. The SPOT installer normally "
|
|
229
|
+
"injects this when the plugin is installed; in tests, set it "
|
|
230
|
+
"manually or use FakeKnowledgeClient."
|
|
231
|
+
)
|
|
232
|
+
return cls(
|
|
233
|
+
url=url,
|
|
234
|
+
api_key=os.environ.get("SPOT_INTERNAL_API_KEY"),
|
|
235
|
+
retrieval_limits=getattr(email, "retrieval_limits", None) or {},
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
# ------------------------------------------------------------------ #
|
|
239
|
+
# Write side: providers
|
|
240
|
+
# ------------------------------------------------------------------ #
|
|
241
|
+
|
|
242
|
+
async def upsert(self, doc: KnowledgeDocument) -> None:
|
|
243
|
+
"""Insert or replace a single document."""
|
|
244
|
+
await self._post("/upsert", json=doc.model_dump(mode="json"), auth=True)
|
|
245
|
+
|
|
246
|
+
async def bulk_upsert(self, docs: list[KnowledgeDocument]) -> None:
|
|
247
|
+
"""Insert or replace many documents in one round-trip."""
|
|
248
|
+
if not docs:
|
|
249
|
+
return
|
|
250
|
+
payload = {"documents": [d.model_dump(mode="json") for d in docs]}
|
|
251
|
+
await self._post("/bulk-upsert", json=payload, auth=True)
|
|
252
|
+
|
|
253
|
+
async def delete(self, doc_id: str) -> None:
|
|
254
|
+
"""Delete a document by id."""
|
|
255
|
+
await self._delete(f"/documents/{doc_id}", auth=True)
|
|
256
|
+
|
|
257
|
+
# ------------------------------------------------------------------ #
|
|
258
|
+
# Read side: analyzers
|
|
259
|
+
# ------------------------------------------------------------------ #
|
|
260
|
+
|
|
261
|
+
async def fetch(
|
|
262
|
+
self,
|
|
263
|
+
tags: str | None = None,
|
|
264
|
+
text: str | None = None,
|
|
265
|
+
top_k: int = 5,
|
|
266
|
+
min_score: float = 0.0,
|
|
267
|
+
strategy: str = "similarity",
|
|
268
|
+
) -> list[KnowledgeDocument]:
|
|
269
|
+
"""Semantic vector search with tag-expression filter.
|
|
270
|
+
|
|
271
|
+
The store embeds ``text`` once (cached by content hash), runs the
|
|
272
|
+
tag filter, returns top-K by cosine similarity (or re-ranked for
|
|
273
|
+
diversity if ``strategy='diverse'``).
|
|
274
|
+
|
|
275
|
+
Args:
|
|
276
|
+
tags: Tag expression — ``"a"``, ``"a+b"`` (AND),
|
|
277
|
+
``"a|b"`` (OR), ``"a+b|c"`` (precedence: AND > OR).
|
|
278
|
+
``None`` / ``""`` means no tag filter.
|
|
279
|
+
text: Free-text query. Required.
|
|
280
|
+
top_k: Number of documents to return.
|
|
281
|
+
min_score: Minimum cosine similarity (0.0 – 1.0).
|
|
282
|
+
strategy: ``"similarity"`` (default) or ``"diverse"``.
|
|
283
|
+
|
|
284
|
+
Returns:
|
|
285
|
+
Documents sorted by ``score`` descending.
|
|
286
|
+
"""
|
|
287
|
+
if text is None or not text.strip():
|
|
288
|
+
raise ValueError("fetch() requires a non-empty 'text' argument")
|
|
289
|
+
|
|
290
|
+
# Apply operator-policy caps (caps only — never raise the request).
|
|
291
|
+
max_top_k = self._limits.get("max_top_k")
|
|
292
|
+
if max_top_k is not None and top_k > max_top_k:
|
|
293
|
+
top_k = max_top_k
|
|
294
|
+
floor = self._limits.get("min_score_floor")
|
|
295
|
+
if floor is not None and min_score < floor:
|
|
296
|
+
min_score = floor
|
|
297
|
+
|
|
298
|
+
payload = {
|
|
299
|
+
"text": text,
|
|
300
|
+
"tags": tags or "",
|
|
301
|
+
"top_k": top_k,
|
|
302
|
+
"min_score": min_score,
|
|
303
|
+
"strategy": strategy,
|
|
304
|
+
}
|
|
305
|
+
data = await self._post("/query", json=payload, auth=False)
|
|
306
|
+
docs_raw = data.get("documents", [])
|
|
307
|
+
return [KnowledgeDocument(**d) for d in docs_raw]
|
|
308
|
+
|
|
309
|
+
# ------------------------------------------------------------------ #
|
|
310
|
+
# HTTP helpers
|
|
311
|
+
# ------------------------------------------------------------------ #
|
|
312
|
+
|
|
313
|
+
def _headers(self, auth: bool) -> dict[str, str]:
|
|
314
|
+
h: dict[str, str] = {"Content-Type": "application/json"}
|
|
315
|
+
if auth and self.api_key:
|
|
316
|
+
h["X-Internal-API-Key"] = self.api_key
|
|
317
|
+
return h
|
|
318
|
+
|
|
319
|
+
async def _post(
|
|
320
|
+
self, path: str, *, json: dict[str, Any], auth: bool
|
|
321
|
+
) -> dict[str, Any]:
|
|
322
|
+
async with httpx.AsyncClient(timeout=self.timeout) as c:
|
|
323
|
+
r = await c.post(self.url + path, json=json, headers=self._headers(auth))
|
|
324
|
+
r.raise_for_status()
|
|
325
|
+
if r.status_code == 204 or not r.content:
|
|
326
|
+
return {}
|
|
327
|
+
return cast("dict[str, Any]", r.json())
|
|
328
|
+
|
|
329
|
+
async def _delete(self, path: str, *, auth: bool) -> None:
|
|
330
|
+
async with httpx.AsyncClient(timeout=self.timeout) as c:
|
|
331
|
+
r = await c.delete(self.url + path, headers=self._headers(auth))
|
|
332
|
+
r.raise_for_status()
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def content_hash(text: str) -> str:
|
|
336
|
+
"""Stable content hash, used as a cache key for embeddings.
|
|
337
|
+
|
|
338
|
+
Exposed for tests and for providers that want to compute IDs from
|
|
339
|
+
content when no natural key exists.
|
|
340
|
+
"""
|
|
341
|
+
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Well-known shared-vocabulary tag constants for the Knowledge Store.
|
|
2
|
+
|
|
3
|
+
Tags are plain strings on ``KnowledgeDocument.tags``; these constants
|
|
4
|
+
encode the recommended common vocabulary so providers and analyzers
|
|
5
|
+
agree on the same labels. External developers may introduce private
|
|
6
|
+
tags (e.g. ``"acme:jira_issue"``) without changing the SDK.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class KnowledgeTag:
|
|
13
|
+
"""Common tag constants. Use these for interoperability across plugins."""
|
|
14
|
+
|
|
15
|
+
# ----- "Kind" tags: what is this document? -----
|
|
16
|
+
EMPLOYEE = "employee"
|
|
17
|
+
WIKI_PAGE = "wiki_page"
|
|
18
|
+
POLICY = "policy"
|
|
19
|
+
ORG_CHART = "org_chart"
|
|
20
|
+
INCIDENT_REPORT = "incident_report"
|
|
21
|
+
TICKET = "ticket"
|
|
22
|
+
THREAT_REPORT = "threat_report"
|
|
23
|
+
SENDER_HISTORY = "sender_history"
|
|
24
|
+
|
|
25
|
+
# ----- "Subset" tags: who/what is this about? -----
|
|
26
|
+
EXECUTIVE = "executive"
|
|
27
|
+
FINANCE = "finance"
|
|
28
|
+
ENGINEERING = "engineering"
|
|
29
|
+
HR = "hr"
|
|
30
|
+
PUBLIC = "public"
|
|
31
|
+
INTERNAL = "internal"
|
spot_sdk/logging.py
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Standard logging configuration for SPOT platform.
|
|
3
|
+
|
|
4
|
+
Uses Python's built-in logging with dictConfig.
|
|
5
|
+
Simple, configurable via environment variables.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
import logging.config
|
|
12
|
+
import os
|
|
13
|
+
from typing import Any, Optional
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def get_log_level() -> str:
|
|
17
|
+
"""Get log level from environment, default to INFO."""
|
|
18
|
+
return os.getenv("LOG_LEVEL", "INFO").upper()
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def get_log_format() -> str:
|
|
22
|
+
"""Get log format from environment, default to simple."""
|
|
23
|
+
return os.getenv("LOG_FORMAT", "simple").lower()
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def get_logging_config(
|
|
27
|
+
log_level: Optional[str] = None, log_format: Optional[str] = None
|
|
28
|
+
) -> dict[str, Any]:
|
|
29
|
+
"""
|
|
30
|
+
Get logging configuration dictionary for dictConfig.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
log_level: Log level (DEBUG, INFO, WARNING, ERROR, CRITICAL)
|
|
34
|
+
log_format: Format style (simple, detailed, json)
|
|
35
|
+
|
|
36
|
+
Returns:
|
|
37
|
+
Dictionary suitable for logging.config.dictConfig
|
|
38
|
+
"""
|
|
39
|
+
level = log_level or get_log_level()
|
|
40
|
+
fmt = log_format or get_log_format()
|
|
41
|
+
|
|
42
|
+
# Choose format string
|
|
43
|
+
if fmt == "json":
|
|
44
|
+
format_str = '{"timestamp": "%(asctime)s", "level": "%(levelname)s", "logger": "%(name)s", "message": "%(message)s"}'
|
|
45
|
+
elif fmt == "detailed":
|
|
46
|
+
format_str = "%(asctime)s - %(name)-20s - %(levelname)-8s - [%(filename)s:%(lineno)d] - %(message)s"
|
|
47
|
+
else: # simple
|
|
48
|
+
format_str = "%(asctime)s - %(name)s - %(levelname)s - %(message)s"
|
|
49
|
+
|
|
50
|
+
config = {
|
|
51
|
+
"version": 1,
|
|
52
|
+
"disable_existing_loggers": False,
|
|
53
|
+
"formatters": {
|
|
54
|
+
"default": {
|
|
55
|
+
"format": format_str,
|
|
56
|
+
"datefmt": "%Y-%m-%d %H:%M:%S" if fmt == "detailed" else "%H:%M:%S",
|
|
57
|
+
}
|
|
58
|
+
},
|
|
59
|
+
"handlers": {
|
|
60
|
+
"console": {
|
|
61
|
+
"class": "logging.StreamHandler",
|
|
62
|
+
"formatter": "default",
|
|
63
|
+
"stream": "ext://sys.stdout",
|
|
64
|
+
}
|
|
65
|
+
},
|
|
66
|
+
"root": {"level": level, "handlers": ["console"]},
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
# Add file handler if LOG_FILE is specified
|
|
70
|
+
log_file = os.getenv("LOG_FILE")
|
|
71
|
+
if log_file:
|
|
72
|
+
config["handlers"]["file"] = { # type: ignore[index]
|
|
73
|
+
"class": "logging.FileHandler",
|
|
74
|
+
"formatter": "default",
|
|
75
|
+
"filename": log_file,
|
|
76
|
+
"mode": "a",
|
|
77
|
+
}
|
|
78
|
+
config["root"]["handlers"].append("file") # type: ignore[index]
|
|
79
|
+
|
|
80
|
+
return config
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def configure_logging(
|
|
84
|
+
log_level: Optional[str] = None, log_format: Optional[str] = None
|
|
85
|
+
) -> None:
|
|
86
|
+
"""
|
|
87
|
+
Configure logging for the application.
|
|
88
|
+
|
|
89
|
+
Call this once during application startup.
|
|
90
|
+
|
|
91
|
+
Args:
|
|
92
|
+
log_level: Log level (default: from LOG_LEVEL env var or INFO)
|
|
93
|
+
log_format: Format style (default: from LOG_FORMAT env var or simple)
|
|
94
|
+
|
|
95
|
+
Environment Variables:
|
|
96
|
+
LOG_LEVEL: DEBUG, INFO, WARNING, ERROR, CRITICAL (default: INFO)
|
|
97
|
+
LOG_FORMAT: json, simple, detailed (default: simple)
|
|
98
|
+
LOG_FILE: Path to log file (optional, logs to console if not set)
|
|
99
|
+
"""
|
|
100
|
+
config = get_logging_config(log_level, log_format)
|
|
101
|
+
logging.config.dictConfig(config)
|
|
102
|
+
|
|
103
|
+
# Log configuration
|
|
104
|
+
logger = logging.getLogger("spot.logging")
|
|
105
|
+
level = log_level or get_log_level()
|
|
106
|
+
fmt = log_format or get_log_format()
|
|
107
|
+
log_file = os.getenv("LOG_FILE")
|
|
108
|
+
|
|
109
|
+
log_targets = "console"
|
|
110
|
+
if log_file:
|
|
111
|
+
log_targets += f" and file ({log_file})"
|
|
112
|
+
|
|
113
|
+
logger.info(
|
|
114
|
+
f"Logging configured - Level: {level}, Format: {fmt}, Output: {log_targets}"
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def get_logger(name: str) -> logging.Logger:
|
|
119
|
+
"""
|
|
120
|
+
Get a logger instance for the given name.
|
|
121
|
+
|
|
122
|
+
Args:
|
|
123
|
+
name: Logger name, typically __name__
|
|
124
|
+
|
|
125
|
+
Returns:
|
|
126
|
+
Standard Python logger
|
|
127
|
+
|
|
128
|
+
Example:
|
|
129
|
+
logger = get_logger(__name__)
|
|
130
|
+
logger.info("Service started")
|
|
131
|
+
logger.debug("Processing email", extra={"email_id": "123"})
|
|
132
|
+
"""
|
|
133
|
+
return logging.getLogger(name)
|
spot_sdk/ollama.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Protocol for Ollama-compatible LLM clients and shared response model."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Protocol, runtime_checkable
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel, Field
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@runtime_checkable
|
|
11
|
+
class OllamaClientProtocol(Protocol):
|
|
12
|
+
"""Protocol defining the interface for Ollama-compatible clients.
|
|
13
|
+
|
|
14
|
+
Both real and mock Ollama clients should implement this protocol.
|
|
15
|
+
The protocol captures the minimal shared surface: generating a text
|
|
16
|
+
response from a prompt and (optionally) lifecycle management.
|
|
17
|
+
|
|
18
|
+
Existing client classes already satisfy this protocol structurally --
|
|
19
|
+
no code changes are needed in analyzer-llm or analyzer-fake-llm.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
async def generate_response(self, prompt: str, model: str) -> str:
|
|
23
|
+
"""Generate a text response from the LLM.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
prompt: The prompt to send to the LLM.
|
|
27
|
+
model: Model name to use for generation.
|
|
28
|
+
|
|
29
|
+
Returns:
|
|
30
|
+
Generated text response (typically JSON for analysis prompts).
|
|
31
|
+
"""
|
|
32
|
+
...
|
|
33
|
+
|
|
34
|
+
async def check_connection(self) -> bool:
|
|
35
|
+
"""Check if the Ollama service is reachable.
|
|
36
|
+
|
|
37
|
+
Returns:
|
|
38
|
+
True if the service is reachable, False otherwise.
|
|
39
|
+
"""
|
|
40
|
+
...
|
|
41
|
+
|
|
42
|
+
async def close(self) -> None:
|
|
43
|
+
"""Release any resources held by the client."""
|
|
44
|
+
...
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class LLMResponse(BaseModel):
|
|
48
|
+
"""Schema for validating LLM JSON responses.
|
|
49
|
+
|
|
50
|
+
Shared between analyzer-llm and analyzer-fake-llm so the response
|
|
51
|
+
format is defined in a single place.
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
is_phishing: bool = False
|
|
55
|
+
confidence: float = Field(default=0.0, ge=0.0, le=1.0)
|
|
56
|
+
threat_level: str = "safe"
|
|
57
|
+
explanation: str = "Analysis completed"
|
|
58
|
+
indicators: list[dict[str, Any]] = Field(default_factory=list)
|
spot_sdk/orchestrator.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Orchestrator result models for SPOT platform."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from pydantic import BaseModel, Field
|
|
7
|
+
|
|
8
|
+
from .analyzer import ThreatLevel
|
|
9
|
+
from .results import AnalysisIndicator
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class AnalyzerResult(BaseModel):
|
|
13
|
+
"""Result from a single analyzer execution."""
|
|
14
|
+
|
|
15
|
+
analyzer_id: str = Field(..., description="Analyzer ID")
|
|
16
|
+
analyzer_type: str = Field(..., description="Type of analyzer")
|
|
17
|
+
execution_time_ms: int = Field(..., description="Execution time in milliseconds")
|
|
18
|
+
status: str = Field(..., description="Execution status (success, failed, timeout)")
|
|
19
|
+
is_phishing: bool = Field(..., description="Whether phishing was detected")
|
|
20
|
+
confidence: float = Field(..., description="Confidence score (0.0 to 1.0)")
|
|
21
|
+
threat_level: ThreatLevel = Field(
|
|
22
|
+
default=ThreatLevel.SAFE, description="Threat level from analyzer"
|
|
23
|
+
)
|
|
24
|
+
indicators: list[AnalysisIndicator] = Field(
|
|
25
|
+
default_factory=list, description="Detected indicators"
|
|
26
|
+
)
|
|
27
|
+
analyzer_details: dict[str, Any] = Field(
|
|
28
|
+
default_factory=dict, description="Additional analyzer-specific details"
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class WorkflowStageResult(BaseModel):
|
|
33
|
+
"""Result from a single workflow stage execution."""
|
|
34
|
+
|
|
35
|
+
stage_name: str = Field(..., description="Stage name")
|
|
36
|
+
stage_type: str = Field(..., description="Stage type")
|
|
37
|
+
execution_time_ms: int = Field(
|
|
38
|
+
..., description="Stage execution time in milliseconds"
|
|
39
|
+
)
|
|
40
|
+
analyzer_results: list[AnalyzerResult] = Field(
|
|
41
|
+
default_factory=list, description="Results from all analyzers in this stage"
|
|
42
|
+
)
|
|
43
|
+
weighted_confidence: float = Field(..., description="Aggregated confidence score")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class OrchestrationResult(BaseModel):
|
|
47
|
+
"""Final orchestration result combining all analyzer outputs."""
|
|
48
|
+
|
|
49
|
+
id: str = Field(..., description="Result ID")
|
|
50
|
+
email_id: str = Field(..., description="Email ID")
|
|
51
|
+
workflow_id: str = Field(..., description="Workflow ID used")
|
|
52
|
+
workflow_version: int = Field(..., description="Workflow version")
|
|
53
|
+
is_phishing: bool = Field(..., description="Final phishing determination")
|
|
54
|
+
threat_level: ThreatLevel = Field(..., description="Overall threat level")
|
|
55
|
+
confidence: float = Field(..., description="Final confidence score (0.0 to 1.0)")
|
|
56
|
+
total_execution_time_ms: int = Field(
|
|
57
|
+
..., description="Total execution time in milliseconds"
|
|
58
|
+
)
|
|
59
|
+
analyzed_at: datetime = Field(..., description="Analysis timestamp")
|
|
60
|
+
summary: str = Field(..., description="Human-readable summary")
|
|
61
|
+
recommended_action: str = Field(..., description="Recommended action")
|
|
62
|
+
reasoning: dict[str, Any] = Field(
|
|
63
|
+
default_factory=dict, description="Reasoning behind the decision"
|
|
64
|
+
)
|
|
65
|
+
config_snapshot: dict[str, Any] = Field(
|
|
66
|
+
default_factory=dict, description="Snapshot of configuration used"
|
|
67
|
+
)
|
|
68
|
+
indicators: list[AnalysisIndicator] = Field(
|
|
69
|
+
default_factory=list, description="All detected indicators"
|
|
70
|
+
)
|
spot_sdk/plugin.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Plugin vocabulary shared across the platform.
|
|
2
|
+
|
|
3
|
+
A "plugin" is any pluggable component registered with a SPOT platform
|
|
4
|
+
through the plugin catalog, installer, and sources. Plugins come in two
|
|
5
|
+
kinds today:
|
|
6
|
+
|
|
7
|
+
- ``analyzer``: runs ``POST /internal/analyze`` and produces an
|
|
8
|
+
``AnalysisResult`` (a phishing verdict contributing to aggregation).
|
|
9
|
+
- ``context_provider``: runs ``POST /internal/enrich`` and produces an
|
|
10
|
+
``EnrichmentResult`` whose data populates ``analysis_context`` for
|
|
11
|
+
downstream analyzers.
|
|
12
|
+
|
|
13
|
+
The ``PluginKind`` value is surfaced by the platform catalog (derived
|
|
14
|
+
from OCI image labels) so the installer can route a plugin to the
|
|
15
|
+
correct section of ``spot.yaml``.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from enum import Enum
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class PluginKind(str, Enum):
|
|
24
|
+
"""Discriminator for pluggable SPOT components."""
|
|
25
|
+
|
|
26
|
+
ANALYZER = "analyzer"
|
|
27
|
+
CONTEXT_PROVIDER = "context_provider"
|
|
28
|
+
|
|
29
|
+
def __str__(self) -> str: # pragma: no cover - trivial
|
|
30
|
+
return self.value
|