spot-sdk-python 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spot_sdk_python-1.0.0/PKG-INFO +353 -0
- spot_sdk_python-1.0.0/README.md +331 -0
- spot_sdk_python-1.0.0/pyproject.toml +32 -0
- spot_sdk_python-1.0.0/spot_sdk/__init__.py +86 -0
- spot_sdk_python-1.0.0/spot_sdk/analysis_context.py +99 -0
- spot_sdk_python-1.0.0/spot_sdk/analyzer.py +106 -0
- spot_sdk_python-1.0.0/spot_sdk/analyzer_base.py +316 -0
- spot_sdk_python-1.0.0/spot_sdk/api_gateway.py +271 -0
- spot_sdk_python-1.0.0/spot_sdk/config.py +46 -0
- spot_sdk_python-1.0.0/spot_sdk/config_client.py +203 -0
- spot_sdk_python-1.0.0/spot_sdk/config_helpers.py +25 -0
- spot_sdk_python-1.0.0/spot_sdk/email.py +136 -0
- spot_sdk_python-1.0.0/spot_sdk/errors.py +24 -0
- spot_sdk_python-1.0.0/spot_sdk/knowledge.py +341 -0
- spot_sdk_python-1.0.0/spot_sdk/knowledge_tags.py +31 -0
- spot_sdk_python-1.0.0/spot_sdk/logging.py +133 -0
- spot_sdk_python-1.0.0/spot_sdk/ollama.py +58 -0
- spot_sdk_python-1.0.0/spot_sdk/orchestrator.py +70 -0
- spot_sdk_python-1.0.0/spot_sdk/plugin.py +30 -0
- spot_sdk_python-1.0.0/spot_sdk/results.py +129 -0
- spot_sdk_python-1.0.0/spot_sdk/settings_schema.py +56 -0
- spot_sdk_python-1.0.0/spot_sdk/testing/README.md +83 -0
- spot_sdk_python-1.0.0/spot_sdk/testing/__init__.py +22 -0
- spot_sdk_python-1.0.0/spot_sdk/testing/factories.py +177 -0
- spot_sdk_python-1.0.0/spot_sdk/testing/fake_knowledge_client.py +105 -0
- spot_sdk_python-1.0.0/spot_sdk/threat_levels.py +33 -0
- spot_sdk_python-1.0.0/spot_sdk/workflow.py +139 -0
|
@@ -0,0 +1,353 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: spot-sdk-python
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Python SDK for SPOT platform - API contracts, models, and utilities
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Author: SPOT Project
|
|
7
|
+
Author-email: spot@sonn.lu
|
|
8
|
+
Requires-Python: >=3.11,<4.0
|
|
9
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
15
|
+
Requires-Dist: httpx (>=0.25.0,<0.26.0)
|
|
16
|
+
Requires-Dist: pydantic-settings (>=2.0.0,<3.0.0)
|
|
17
|
+
Requires-Dist: pydantic[email] (>=2.0.0,<3.0.0)
|
|
18
|
+
Requires-Dist: typing-extensions (>=4.8.0,<5.0.0)
|
|
19
|
+
Project-URL: Homepage, https://sonn.lu
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
|
|
22
|
+
# SPOT Contracts Python SDK
|
|
23
|
+
|
|
24
|
+
Python SDK generated from SPOT Platform OpenAPI specifications.
|
|
25
|
+
|
|
26
|
+
## Installation
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install spot-sdk-python
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
## Usage
|
|
33
|
+
|
|
34
|
+
### Basic Usage
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
from spot_sdk.analyzer import Email, EmailHeader, AnalysisResult
|
|
38
|
+
from spot_sdk.api_gateway import AnalysisRequest
|
|
39
|
+
|
|
40
|
+
# Create email header
|
|
41
|
+
header = EmailHeader(
|
|
42
|
+
message_id="msg-001",
|
|
43
|
+
subject="Test Email",
|
|
44
|
+
sender="sender@example.com",
|
|
45
|
+
recipients=["recipient@example.com"],
|
|
46
|
+
date="2024-01-01T12:00:00Z"
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
# Create email object
|
|
50
|
+
email = Email(
|
|
51
|
+
id="12345",
|
|
52
|
+
headers=header,
|
|
53
|
+
body_text="This is the email content"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# Use with API Gateway
|
|
57
|
+
request = AnalysisRequest(email=email.dict())
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
### Pydantic Models
|
|
61
|
+
|
|
62
|
+
All models are Pydantic v2 BaseModel instances with:
|
|
63
|
+
|
|
64
|
+
- **Type validation** - Automatic validation of field types
|
|
65
|
+
- **Serialization** - `.dict()` and `.json()` methods
|
|
66
|
+
- **Field descriptions** - Documentation from OpenAPI specs
|
|
67
|
+
- **IDE support** - Full type hints and autocompletion
|
|
68
|
+
|
|
69
|
+
### Example: Email Analysis
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
import httpx
|
|
73
|
+
from spot_sdk.analyzer import Email
|
|
74
|
+
from spot_sdk.api_gateway import AnalysisRequest
|
|
75
|
+
|
|
76
|
+
async def analyze_email(email: Email) -> dict:
|
|
77
|
+
async with httpx.AsyncClient() as client:
|
|
78
|
+
request = AnalysisRequest(email=email.dict())
|
|
79
|
+
response = await client.post(
|
|
80
|
+
"http://localhost:8001/api/v1/analyze",
|
|
81
|
+
json=request.dict()
|
|
82
|
+
)
|
|
83
|
+
return response.json()
|
|
84
|
+
|
|
85
|
+
# Usage
|
|
86
|
+
email = Email(
|
|
87
|
+
id="sample-123",
|
|
88
|
+
headers={
|
|
89
|
+
"message_id": "msg-sample",
|
|
90
|
+
"subject": "Urgent: Action Required",
|
|
91
|
+
"sender": "suspicious@example.com",
|
|
92
|
+
"recipients": ["target@company.com"],
|
|
93
|
+
"date": "2024-01-15T10:30:00Z"
|
|
94
|
+
},
|
|
95
|
+
body_text="Please click this link immediately..."
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
result = await analyze_email(email)
|
|
99
|
+
print(f"Phishing detected: {result['is_phishing']}")
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
### Accessing Previous Stage Results (`analysis_context`)
|
|
103
|
+
|
|
104
|
+
The orchestrator automatically populates `Email.analysis_context` with the results of all previously-completed stages before calling each analyzer. No workflow configuration is required -- analyzers can read whatever they need.
|
|
105
|
+
|
|
106
|
+
**Structure:**
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
email.analysis_context = {
|
|
110
|
+
"<stage-name>": {
|
|
111
|
+
"providers": {
|
|
112
|
+
"<provider-id>": { ...free-form data... }
|
|
113
|
+
},
|
|
114
|
+
"analyzers": {
|
|
115
|
+
"<analyzer-id>": { ...AnalyzerResult fields... }
|
|
116
|
+
}
|
|
117
|
+
},
|
|
118
|
+
...
|
|
119
|
+
}
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
- Top-level keys are stage names
|
|
123
|
+
- Each stage has a `providers` dict (populated by Context Providers) and an `analyzers` dict
|
|
124
|
+
- Analyzer values expose all `AnalyzerResult` fields (`is_phishing`, `confidence`, `threat_level`, `indicators`, `analyzer_details`, ...)
|
|
125
|
+
- Only stages that have completed before the current analyzer runs are present
|
|
126
|
+
|
|
127
|
+
**Recommended: use the `email.ctx` helper**
|
|
128
|
+
|
|
129
|
+
The SDK provides an `AnalysisContextReader` (via the `Email.ctx` property) with ergonomic accessors so you don't have to write manual dict traversal:
|
|
130
|
+
|
|
131
|
+
```python
|
|
132
|
+
@app.post("/internal/analyze")
|
|
133
|
+
async def analyze_email(email: Email) -> AnalysisResult:
|
|
134
|
+
ctx = email.ctx # AnalysisContextReader
|
|
135
|
+
|
|
136
|
+
# Get a specific analyzer from any stage (first match)
|
|
137
|
+
nlp = ctx.analyzer("analyzer-nlp")
|
|
138
|
+
if nlp:
|
|
139
|
+
confidence = nlp["confidence"]
|
|
140
|
+
|
|
141
|
+
# Get a specific analyzer from a specific stage
|
|
142
|
+
ml = ctx.analyzer("analyzer-ml", stage="parallel-analysis")
|
|
143
|
+
|
|
144
|
+
# Get a provider's data
|
|
145
|
+
dept = ctx.provider("employee-dir")
|
|
146
|
+
if dept:
|
|
147
|
+
role = dept.get("role")
|
|
148
|
+
|
|
149
|
+
# Iterate all analyzers in a stage
|
|
150
|
+
for aid, result in ctx.analyzers_in("parallel-analysis").items():
|
|
151
|
+
print(aid, result["confidence"])
|
|
152
|
+
|
|
153
|
+
# Listing helpers
|
|
154
|
+
ctx.stages() # ['enrichment', 'parallel-analysis']
|
|
155
|
+
ctx.analyzer_ids() # ['analyzer-nlp', 'analyzer-ml']
|
|
156
|
+
ctx.provider_ids() # ['employee-dir', 'threat-feed']
|
|
157
|
+
ctx.analyzer_ids_in("parallel-analysis") # stage-scoped
|
|
158
|
+
ctx.provider_ids_in("enrichment")
|
|
159
|
+
ctx.has_stage("parallel-analysis") # True/False
|
|
160
|
+
|
|
161
|
+
# Bulk accessors: {stage: {id: data, ...}, ...}
|
|
162
|
+
ctx.all_analyzers()
|
|
163
|
+
ctx.all_providers()
|
|
164
|
+
...
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
All single accessors (`analyzer()`, `provider()`) return `None` when not found. Bulk accessors return empty dicts. Nothing raises -- missing stages or IDs are treated as normal.
|
|
168
|
+
|
|
169
|
+
**Raw access (if you prefer)**
|
|
170
|
+
|
|
171
|
+
You can also read `email.analysis_context` directly as a plain dict:
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
prev = email.analysis_context.get("parallel-analysis", {})
|
|
175
|
+
nlp = prev.get("analyzers", {}).get("analyzer-nlp")
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
Analyzers that don't need previous results can safely ignore this field -- it defaults to an empty dict.
|
|
179
|
+
|
|
180
|
+
## Plugins
|
|
181
|
+
|
|
182
|
+
SPOT's umbrella vocabulary for pluggable components is **plugin**.
|
|
183
|
+
``PluginKind`` discriminates between the two kinds today:
|
|
184
|
+
|
|
185
|
+
```python
|
|
186
|
+
from spot_sdk import PluginKind
|
|
187
|
+
|
|
188
|
+
PluginKind.ANALYZER # analyzer kind
|
|
189
|
+
PluginKind.CONTEXT_PROVIDER # context provider kind
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
- **Analyzers** run ``POST /internal/analyze`` and return an
|
|
193
|
+
``AnalysisResult`` (a phishing verdict contributing to aggregation).
|
|
194
|
+
- **Context providers** run ``POST /internal/enrich`` and return an
|
|
195
|
+
``EnrichmentResult`` whose data populates ``analysis_context`` for
|
|
196
|
+
downstream analyzers.
|
|
197
|
+
|
|
198
|
+
Plugins are discovered from OCI image labels prefixed ``spot.plugin.*``
|
|
199
|
+
(see the platform documentation for the label contract) and installed
|
|
200
|
+
into the platform via the ``/api/v1/plugins/*`` and
|
|
201
|
+
``/api/v1/config/plugin/{kind}`` APIs.
|
|
202
|
+
|
|
203
|
+
## Context Providers
|
|
204
|
+
|
|
205
|
+
Context providers enrich an email with organizational data (employee directory,
|
|
206
|
+
sender history, threat intelligence, knowledge bases, ...) before analyzers run.
|
|
207
|
+
They are defined per workflow stage and run in parallel, just like analyzers.
|
|
208
|
+
|
|
209
|
+
Each provider exposes a single HTTP endpoint:
|
|
210
|
+
|
|
211
|
+
```
|
|
212
|
+
POST /internal/enrich
|
|
213
|
+
Content-Type: application/json
|
|
214
|
+
|
|
215
|
+
Input: Email (same model analyzers receive)
|
|
216
|
+
Output: EnrichmentResult
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
### Implementing a provider
|
|
220
|
+
|
|
221
|
+
```python
|
|
222
|
+
from fastapi import FastAPI
|
|
223
|
+
from spot_sdk import Email, EnrichmentResult
|
|
224
|
+
|
|
225
|
+
app = FastAPI()
|
|
226
|
+
|
|
227
|
+
@app.post("/internal/enrich")
|
|
228
|
+
async def enrich(email: Email) -> EnrichmentResult:
|
|
229
|
+
sender = email.headers.sender
|
|
230
|
+
return EnrichmentResult(
|
|
231
|
+
provider_id="employee-dir",
|
|
232
|
+
data={
|
|
233
|
+
"sender_known": True,
|
|
234
|
+
"sender_department": "Finance",
|
|
235
|
+
"sender_role": "CFO",
|
|
236
|
+
},
|
|
237
|
+
source="company-ldap",
|
|
238
|
+
confidence=0.95,
|
|
239
|
+
)
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
The orchestrator merges each provider's `data` dict into
|
|
243
|
+
`email.analysis_context["<stage-name>"]["providers"]["<provider-id>"]`,
|
|
244
|
+
where downstream analyzers can read it via `email.ctx.provider(...)`.
|
|
245
|
+
|
|
246
|
+
## Knowledge Store
|
|
247
|
+
|
|
248
|
+
The Knowledge Store is the platform's RAG layer: context providers
|
|
249
|
+
deposit tagged documents, analyzers fetch them on demand. Neither side
|
|
250
|
+
references the other — they share only the tag vocabulary.
|
|
251
|
+
|
|
252
|
+
```python
|
|
253
|
+
from spot_sdk import KnowledgeClient, KnowledgeDocument, KnowledgeTag, chunk_text
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
### Ingestion (providers)
|
|
257
|
+
|
|
258
|
+
Upsert is idempotent on `id`. Use stable deterministic ids so re-syncs
|
|
259
|
+
replace rather than duplicate.
|
|
260
|
+
|
|
261
|
+
```python
|
|
262
|
+
kb = KnowledgeClient(
|
|
263
|
+
url=os.environ["SPOT_KNOWLEDGE_URL"], # injected by installer
|
|
264
|
+
api_key=os.environ["SPOT_INTERNAL_API_KEY"], # idem
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
await kb.bulk_upsert([
|
|
268
|
+
KnowledgeDocument(
|
|
269
|
+
id=f"employee:{e.email}",
|
|
270
|
+
content=f"{e.name}, {e.title}, {e.department}",
|
|
271
|
+
tags=[KnowledgeTag.EMPLOYEE, *([KnowledgeTag.EXECUTIVE] if e.is_exec else [])],
|
|
272
|
+
metadata={"email": e.email, "title": e.title},
|
|
273
|
+
source="provider-employee-dir",
|
|
274
|
+
)
|
|
275
|
+
for e in employees
|
|
276
|
+
])
|
|
277
|
+
```
|
|
278
|
+
|
|
279
|
+
For long content use `chunk_text(body, max_chars=2000, overlap=200)`
|
|
280
|
+
and upsert each chunk with `metadata["parent_id"]` set to the source
|
|
281
|
+
doc id.
|
|
282
|
+
|
|
283
|
+
### Consumption (analyzers)
|
|
284
|
+
|
|
285
|
+
```python
|
|
286
|
+
kb = KnowledgeClient.for_analysis(email) # reads SPOT_KNOWLEDGE_URL +
|
|
287
|
+
# email.retrieval_limits
|
|
288
|
+
docs = await kb.fetch(tags="employee+executive", text=email.sender, top_k=3)
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
Tag-expression syntax: `a`, `a+b` (AND), `a|b` (OR), `a+b|c`
|
|
292
|
+
(`(a AND b) OR c`). `+` binds tighter than `|`. Empty or `None` means
|
|
293
|
+
no tag filter.
|
|
294
|
+
|
|
295
|
+
Workflow-declared `retrieval_limits` (stage-level caps on `top_k` and
|
|
296
|
+
`min_score`) are enforced transparently by `for_analysis()` — the
|
|
297
|
+
analyzer doesn't need to know about them.
|
|
298
|
+
|
|
299
|
+
### Testing
|
|
300
|
+
|
|
301
|
+
```python
|
|
302
|
+
from spot_sdk.testing.fake_knowledge_client import FakeKnowledgeClient
|
|
303
|
+
|
|
304
|
+
kb = FakeKnowledgeClient()
|
|
305
|
+
await kb.upsert(KnowledgeDocument(id="e:a", content="Alice", tags=["employee"]))
|
|
306
|
+
assert (await kb.fetch(tags="employee", text="Alice"))[0].id == "e:a"
|
|
307
|
+
```
|
|
308
|
+
|
|
309
|
+
## Available Models
|
|
310
|
+
|
|
311
|
+
### Analyzer Service (`spot_sdk.analyzer`)
|
|
312
|
+
- `Email` - Email data structure
|
|
313
|
+
- `EmailHeader` - Email header fields
|
|
314
|
+
- `Attachment` - Email attachment
|
|
315
|
+
- `AnalysisResult` - Analysis results
|
|
316
|
+
- `AnalysisIndicator` - Phishing indicators
|
|
317
|
+
- `ThreatLevel` - Threat level enum
|
|
318
|
+
|
|
319
|
+
### API Gateway (`spot_sdk.api_gateway`)
|
|
320
|
+
- `AnalysisRequest` - Analysis request
|
|
321
|
+
- `AnalysisResponse` - Analysis response
|
|
322
|
+
- `ConfigRequest` - Configuration request
|
|
323
|
+
- `ConfigResponse` - Configuration response
|
|
324
|
+
|
|
325
|
+
### Workflow (`spot_sdk.workflow`)
|
|
326
|
+
- `Workflow`, `WorkflowStage`
|
|
327
|
+
- `AnalyzerConfig`, `ContextProviderConfig`
|
|
328
|
+
- `RetryConfig`, `FailureStrategy`
|
|
329
|
+
|
|
330
|
+
### Plugin vocabulary (`spot_sdk.plugin`)
|
|
331
|
+
- `PluginKind` - `ANALYZER` | `CONTEXT_PROVIDER`
|
|
332
|
+
|
|
333
|
+
### Enrichment (`spot_sdk.enrichment`)
|
|
334
|
+
- `EnrichmentResult` - Returned by context providers from `/internal/enrich`
|
|
335
|
+
|
|
336
|
+
## Requirements
|
|
337
|
+
|
|
338
|
+
- Python 3.11+
|
|
339
|
+
- pydantic >= 2.0.0
|
|
340
|
+
- httpx >= 0.25.0
|
|
341
|
+
- typing-extensions >= 4.8.0
|
|
342
|
+
|
|
343
|
+
## Development
|
|
344
|
+
|
|
345
|
+
This SDK is automatically generated from OpenAPI specifications.
|
|
346
|
+
Do not modify generated files directly - update the OpenAPI specs instead.
|
|
347
|
+
|
|
348
|
+
## Version
|
|
349
|
+
|
|
350
|
+
Current version: 1.0.0
|
|
351
|
+
|
|
352
|
+
Generated from SPOT Contracts repository.
|
|
353
|
+
|
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
# SPOT Contracts Python SDK
|
|
2
|
+
|
|
3
|
+
Python SDK generated from SPOT Platform OpenAPI specifications.
|
|
4
|
+
|
|
5
|
+
## Installation
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install spot-sdk-python
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Usage
|
|
12
|
+
|
|
13
|
+
### Basic Usage
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
from spot_sdk.analyzer import Email, EmailHeader, AnalysisResult
|
|
17
|
+
from spot_sdk.api_gateway import AnalysisRequest
|
|
18
|
+
|
|
19
|
+
# Create email header
|
|
20
|
+
header = EmailHeader(
|
|
21
|
+
message_id="msg-001",
|
|
22
|
+
subject="Test Email",
|
|
23
|
+
sender="sender@example.com",
|
|
24
|
+
recipients=["recipient@example.com"],
|
|
25
|
+
date="2024-01-01T12:00:00Z"
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# Create email object
|
|
29
|
+
email = Email(
|
|
30
|
+
id="12345",
|
|
31
|
+
headers=header,
|
|
32
|
+
body_text="This is the email content"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
# Use with API Gateway
|
|
36
|
+
request = AnalysisRequest(email=email.dict())
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
### Pydantic Models
|
|
40
|
+
|
|
41
|
+
All models are Pydantic v2 BaseModel instances with:
|
|
42
|
+
|
|
43
|
+
- **Type validation** - Automatic validation of field types
|
|
44
|
+
- **Serialization** - `.dict()` and `.json()` methods
|
|
45
|
+
- **Field descriptions** - Documentation from OpenAPI specs
|
|
46
|
+
- **IDE support** - Full type hints and autocompletion
|
|
47
|
+
|
|
48
|
+
### Example: Email Analysis
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
import httpx
|
|
52
|
+
from spot_sdk.analyzer import Email
|
|
53
|
+
from spot_sdk.api_gateway import AnalysisRequest
|
|
54
|
+
|
|
55
|
+
async def analyze_email(email: Email) -> dict:
|
|
56
|
+
async with httpx.AsyncClient() as client:
|
|
57
|
+
request = AnalysisRequest(email=email.dict())
|
|
58
|
+
response = await client.post(
|
|
59
|
+
"http://localhost:8001/api/v1/analyze",
|
|
60
|
+
json=request.dict()
|
|
61
|
+
)
|
|
62
|
+
return response.json()
|
|
63
|
+
|
|
64
|
+
# Usage
|
|
65
|
+
email = Email(
|
|
66
|
+
id="sample-123",
|
|
67
|
+
headers={
|
|
68
|
+
"message_id": "msg-sample",
|
|
69
|
+
"subject": "Urgent: Action Required",
|
|
70
|
+
"sender": "suspicious@example.com",
|
|
71
|
+
"recipients": ["target@company.com"],
|
|
72
|
+
"date": "2024-01-15T10:30:00Z"
|
|
73
|
+
},
|
|
74
|
+
body_text="Please click this link immediately..."
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
result = await analyze_email(email)
|
|
78
|
+
print(f"Phishing detected: {result['is_phishing']}")
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
### Accessing Previous Stage Results (`analysis_context`)
|
|
82
|
+
|
|
83
|
+
The orchestrator automatically populates `Email.analysis_context` with the results of all previously-completed stages before calling each analyzer. No workflow configuration is required -- analyzers can read whatever they need.
|
|
84
|
+
|
|
85
|
+
**Structure:**
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
email.analysis_context = {
|
|
89
|
+
"<stage-name>": {
|
|
90
|
+
"providers": {
|
|
91
|
+
"<provider-id>": { ...free-form data... }
|
|
92
|
+
},
|
|
93
|
+
"analyzers": {
|
|
94
|
+
"<analyzer-id>": { ...AnalyzerResult fields... }
|
|
95
|
+
}
|
|
96
|
+
},
|
|
97
|
+
...
|
|
98
|
+
}
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
- Top-level keys are stage names
|
|
102
|
+
- Each stage has a `providers` dict (populated by Context Providers) and an `analyzers` dict
|
|
103
|
+
- Analyzer values expose all `AnalyzerResult` fields (`is_phishing`, `confidence`, `threat_level`, `indicators`, `analyzer_details`, ...)
|
|
104
|
+
- Only stages that have completed before the current analyzer runs are present
|
|
105
|
+
|
|
106
|
+
**Recommended: use the `email.ctx` helper**
|
|
107
|
+
|
|
108
|
+
The SDK provides an `AnalysisContextReader` (via the `Email.ctx` property) with ergonomic accessors so you don't have to write manual dict traversal:
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
@app.post("/internal/analyze")
|
|
112
|
+
async def analyze_email(email: Email) -> AnalysisResult:
|
|
113
|
+
ctx = email.ctx # AnalysisContextReader
|
|
114
|
+
|
|
115
|
+
# Get a specific analyzer from any stage (first match)
|
|
116
|
+
nlp = ctx.analyzer("analyzer-nlp")
|
|
117
|
+
if nlp:
|
|
118
|
+
confidence = nlp["confidence"]
|
|
119
|
+
|
|
120
|
+
# Get a specific analyzer from a specific stage
|
|
121
|
+
ml = ctx.analyzer("analyzer-ml", stage="parallel-analysis")
|
|
122
|
+
|
|
123
|
+
# Get a provider's data
|
|
124
|
+
dept = ctx.provider("employee-dir")
|
|
125
|
+
if dept:
|
|
126
|
+
role = dept.get("role")
|
|
127
|
+
|
|
128
|
+
# Iterate all analyzers in a stage
|
|
129
|
+
for aid, result in ctx.analyzers_in("parallel-analysis").items():
|
|
130
|
+
print(aid, result["confidence"])
|
|
131
|
+
|
|
132
|
+
# Listing helpers
|
|
133
|
+
ctx.stages() # ['enrichment', 'parallel-analysis']
|
|
134
|
+
ctx.analyzer_ids() # ['analyzer-nlp', 'analyzer-ml']
|
|
135
|
+
ctx.provider_ids() # ['employee-dir', 'threat-feed']
|
|
136
|
+
ctx.analyzer_ids_in("parallel-analysis") # stage-scoped
|
|
137
|
+
ctx.provider_ids_in("enrichment")
|
|
138
|
+
ctx.has_stage("parallel-analysis") # True/False
|
|
139
|
+
|
|
140
|
+
# Bulk accessors: {stage: {id: data, ...}, ...}
|
|
141
|
+
ctx.all_analyzers()
|
|
142
|
+
ctx.all_providers()
|
|
143
|
+
...
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
All single accessors (`analyzer()`, `provider()`) return `None` when not found. Bulk accessors return empty dicts. Nothing raises -- missing stages or IDs are treated as normal.
|
|
147
|
+
|
|
148
|
+
**Raw access (if you prefer)**
|
|
149
|
+
|
|
150
|
+
You can also read `email.analysis_context` directly as a plain dict:
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
prev = email.analysis_context.get("parallel-analysis", {})
|
|
154
|
+
nlp = prev.get("analyzers", {}).get("analyzer-nlp")
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
Analyzers that don't need previous results can safely ignore this field -- it defaults to an empty dict.
|
|
158
|
+
|
|
159
|
+
## Plugins
|
|
160
|
+
|
|
161
|
+
SPOT's umbrella vocabulary for pluggable components is **plugin**.
|
|
162
|
+
``PluginKind`` discriminates between the two kinds today:
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
from spot_sdk import PluginKind
|
|
166
|
+
|
|
167
|
+
PluginKind.ANALYZER # analyzer kind
|
|
168
|
+
PluginKind.CONTEXT_PROVIDER # context provider kind
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
- **Analyzers** run ``POST /internal/analyze`` and return an
|
|
172
|
+
``AnalysisResult`` (a phishing verdict contributing to aggregation).
|
|
173
|
+
- **Context providers** run ``POST /internal/enrich`` and return an
|
|
174
|
+
``EnrichmentResult`` whose data populates ``analysis_context`` for
|
|
175
|
+
downstream analyzers.
|
|
176
|
+
|
|
177
|
+
Plugins are discovered from OCI image labels prefixed ``spot.plugin.*``
|
|
178
|
+
(see the platform documentation for the label contract) and installed
|
|
179
|
+
into the platform via the ``/api/v1/plugins/*`` and
|
|
180
|
+
``/api/v1/config/plugin/{kind}`` APIs.
|
|
181
|
+
|
|
182
|
+
## Context Providers
|
|
183
|
+
|
|
184
|
+
Context providers enrich an email with organizational data (employee directory,
|
|
185
|
+
sender history, threat intelligence, knowledge bases, ...) before analyzers run.
|
|
186
|
+
They are defined per workflow stage and run in parallel, just like analyzers.
|
|
187
|
+
|
|
188
|
+
Each provider exposes a single HTTP endpoint:
|
|
189
|
+
|
|
190
|
+
```
|
|
191
|
+
POST /internal/enrich
|
|
192
|
+
Content-Type: application/json
|
|
193
|
+
|
|
194
|
+
Input: Email (same model analyzers receive)
|
|
195
|
+
Output: EnrichmentResult
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
### Implementing a provider
|
|
199
|
+
|
|
200
|
+
```python
|
|
201
|
+
from fastapi import FastAPI
|
|
202
|
+
from spot_sdk import Email, EnrichmentResult
|
|
203
|
+
|
|
204
|
+
app = FastAPI()
|
|
205
|
+
|
|
206
|
+
@app.post("/internal/enrich")
|
|
207
|
+
async def enrich(email: Email) -> EnrichmentResult:
|
|
208
|
+
sender = email.headers.sender
|
|
209
|
+
return EnrichmentResult(
|
|
210
|
+
provider_id="employee-dir",
|
|
211
|
+
data={
|
|
212
|
+
"sender_known": True,
|
|
213
|
+
"sender_department": "Finance",
|
|
214
|
+
"sender_role": "CFO",
|
|
215
|
+
},
|
|
216
|
+
source="company-ldap",
|
|
217
|
+
confidence=0.95,
|
|
218
|
+
)
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
The orchestrator merges each provider's `data` dict into
|
|
222
|
+
`email.analysis_context["<stage-name>"]["providers"]["<provider-id>"]`,
|
|
223
|
+
where downstream analyzers can read it via `email.ctx.provider(...)`.
|
|
224
|
+
|
|
225
|
+
## Knowledge Store
|
|
226
|
+
|
|
227
|
+
The Knowledge Store is the platform's RAG layer: context providers
|
|
228
|
+
deposit tagged documents, analyzers fetch them on demand. Neither side
|
|
229
|
+
references the other — they share only the tag vocabulary.
|
|
230
|
+
|
|
231
|
+
```python
|
|
232
|
+
from spot_sdk import KnowledgeClient, KnowledgeDocument, KnowledgeTag, chunk_text
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
### Ingestion (providers)
|
|
236
|
+
|
|
237
|
+
Upsert is idempotent on `id`. Use stable deterministic ids so re-syncs
|
|
238
|
+
replace rather than duplicate.
|
|
239
|
+
|
|
240
|
+
```python
|
|
241
|
+
kb = KnowledgeClient(
|
|
242
|
+
url=os.environ["SPOT_KNOWLEDGE_URL"], # injected by installer
|
|
243
|
+
api_key=os.environ["SPOT_INTERNAL_API_KEY"], # idem
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
await kb.bulk_upsert([
|
|
247
|
+
KnowledgeDocument(
|
|
248
|
+
id=f"employee:{e.email}",
|
|
249
|
+
content=f"{e.name}, {e.title}, {e.department}",
|
|
250
|
+
tags=[KnowledgeTag.EMPLOYEE, *([KnowledgeTag.EXECUTIVE] if e.is_exec else [])],
|
|
251
|
+
metadata={"email": e.email, "title": e.title},
|
|
252
|
+
source="provider-employee-dir",
|
|
253
|
+
)
|
|
254
|
+
for e in employees
|
|
255
|
+
])
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
For long content use `chunk_text(body, max_chars=2000, overlap=200)`
|
|
259
|
+
and upsert each chunk with `metadata["parent_id"]` set to the source
|
|
260
|
+
doc id.
|
|
261
|
+
|
|
262
|
+
### Consumption (analyzers)
|
|
263
|
+
|
|
264
|
+
```python
|
|
265
|
+
kb = KnowledgeClient.for_analysis(email) # reads SPOT_KNOWLEDGE_URL +
|
|
266
|
+
# email.retrieval_limits
|
|
267
|
+
docs = await kb.fetch(tags="employee+executive", text=email.sender, top_k=3)
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
Tag-expression syntax: `a`, `a+b` (AND), `a|b` (OR), `a+b|c`
|
|
271
|
+
(`(a AND b) OR c`). `+` binds tighter than `|`. Empty or `None` means
|
|
272
|
+
no tag filter.
|
|
273
|
+
|
|
274
|
+
Workflow-declared `retrieval_limits` (stage-level caps on `top_k` and
|
|
275
|
+
`min_score`) are enforced transparently by `for_analysis()` — the
|
|
276
|
+
analyzer doesn't need to know about them.
|
|
277
|
+
|
|
278
|
+
### Testing
|
|
279
|
+
|
|
280
|
+
```python
|
|
281
|
+
from spot_sdk.testing.fake_knowledge_client import FakeKnowledgeClient
|
|
282
|
+
|
|
283
|
+
kb = FakeKnowledgeClient()
|
|
284
|
+
await kb.upsert(KnowledgeDocument(id="e:a", content="Alice", tags=["employee"]))
|
|
285
|
+
assert (await kb.fetch(tags="employee", text="Alice"))[0].id == "e:a"
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
## Available Models
|
|
289
|
+
|
|
290
|
+
### Analyzer Service (`spot_sdk.analyzer`)
|
|
291
|
+
- `Email` - Email data structure
|
|
292
|
+
- `EmailHeader` - Email header fields
|
|
293
|
+
- `Attachment` - Email attachment
|
|
294
|
+
- `AnalysisResult` - Analysis results
|
|
295
|
+
- `AnalysisIndicator` - Phishing indicators
|
|
296
|
+
- `ThreatLevel` - Threat level enum
|
|
297
|
+
|
|
298
|
+
### API Gateway (`spot_sdk.api_gateway`)
|
|
299
|
+
- `AnalysisRequest` - Analysis request
|
|
300
|
+
- `AnalysisResponse` - Analysis response
|
|
301
|
+
- `ConfigRequest` - Configuration request
|
|
302
|
+
- `ConfigResponse` - Configuration response
|
|
303
|
+
|
|
304
|
+
### Workflow (`spot_sdk.workflow`)
|
|
305
|
+
- `Workflow`, `WorkflowStage`
|
|
306
|
+
- `AnalyzerConfig`, `ContextProviderConfig`
|
|
307
|
+
- `RetryConfig`, `FailureStrategy`
|
|
308
|
+
|
|
309
|
+
### Plugin vocabulary (`spot_sdk.plugin`)
|
|
310
|
+
- `PluginKind` - `ANALYZER` | `CONTEXT_PROVIDER`
|
|
311
|
+
|
|
312
|
+
### Enrichment (`spot_sdk.enrichment`)
|
|
313
|
+
- `EnrichmentResult` - Returned by context providers from `/internal/enrich`
|
|
314
|
+
|
|
315
|
+
## Requirements
|
|
316
|
+
|
|
317
|
+
- Python 3.11+
|
|
318
|
+
- pydantic >= 2.0.0
|
|
319
|
+
- httpx >= 0.25.0
|
|
320
|
+
- typing-extensions >= 4.8.0
|
|
321
|
+
|
|
322
|
+
## Development
|
|
323
|
+
|
|
324
|
+
This SDK is automatically generated from OpenAPI specifications.
|
|
325
|
+
Do not modify generated files directly - update the OpenAPI specs instead.
|
|
326
|
+
|
|
327
|
+
## Version
|
|
328
|
+
|
|
329
|
+
Current version: 1.0.0
|
|
330
|
+
|
|
331
|
+
Generated from SPOT Contracts repository.
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["poetry-core"]
|
|
3
|
+
build-backend = "poetry.core.masonry.api"
|
|
4
|
+
|
|
5
|
+
[tool.poetry]
|
|
6
|
+
name = "spot-sdk-python"
|
|
7
|
+
version = "1.0.0"
|
|
8
|
+
description = "Python SDK for SPOT platform - API contracts, models, and utilities"
|
|
9
|
+
authors = ["SPOT Project <spot@sonn.lu>"]
|
|
10
|
+
license = "Apache-2.0"
|
|
11
|
+
homepage = "https://sonn.lu"
|
|
12
|
+
readme = "README.md"
|
|
13
|
+
packages = [
|
|
14
|
+
{include = "spot_sdk"}
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[tool.poetry.dependencies]
|
|
18
|
+
python = "^3.11"
|
|
19
|
+
pydantic = {extras = ["email"], version = "^2.0.0"}
|
|
20
|
+
pydantic-settings = "^2.0.0"
|
|
21
|
+
httpx = "^0.25.0"
|
|
22
|
+
typing-extensions = "^4.8.0"
|
|
23
|
+
|
|
24
|
+
[tool.poetry.group.dev.dependencies]
|
|
25
|
+
pytest = "^7.4.0"
|
|
26
|
+
pytest-asyncio = "^0.21.0"
|
|
27
|
+
mypy = "^1.5.0"
|
|
28
|
+
|
|
29
|
+
[tool.ruff]
|
|
30
|
+
select = ["E", "F", "I", "N", "W"]
|
|
31
|
+
ignore = ["E501"] # Line too long
|
|
32
|
+
target-version = "py311"
|