spot-sdk-python 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
spot_sdk/workflow.py ADDED
@@ -0,0 +1,139 @@
1
+ """Workflow configuration models for SPOT platform."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from enum import Enum
6
+
7
+ from pydantic import BaseModel, Field
8
+
9
+
10
+ class FailureStrategy(str, Enum):
11
+ """Strategy for handling analyzer failures."""
12
+
13
+ RETRY = "retry"
14
+ SKIP = "skip"
15
+ FAIL = "fail"
16
+ FALLBACK = "fallback"
17
+
18
+
19
+ class RetryConfig(BaseModel):
20
+ """Retry configuration for analyzer execution."""
21
+
22
+ max_attempts: int = Field(
23
+ default=3, ge=1, le=10, description="Maximum number of retry attempts"
24
+ )
25
+ backoff_ms: int = Field(
26
+ default=1000,
27
+ ge=100,
28
+ le=60000,
29
+ description="Initial backoff time in milliseconds",
30
+ )
31
+ max_backoff_ms: int = Field(
32
+ default=10000,
33
+ ge=1000,
34
+ le=300000,
35
+ description="Maximum backoff time in milliseconds",
36
+ )
37
+ exponential_backoff: bool = Field(
38
+ default=True, description="Use exponential backoff"
39
+ )
40
+
41
+
42
+ class AnalyzerConfig(BaseModel):
43
+ """Configuration for a single analyzer in a workflow stage."""
44
+
45
+ id: str = Field(..., description="Analyzer ID")
46
+ weight: float = Field(
47
+ default=1.0, ge=0.0, le=10.0, description="Weight for result aggregation"
48
+ )
49
+ timeout_ms: int = Field(
50
+ default=30000, ge=1000, le=300000, description="Timeout in milliseconds"
51
+ )
52
+ required: bool = Field(
53
+ default=False, description="Whether this analyzer is required"
54
+ )
55
+ failure_strategy: FailureStrategy = Field(
56
+ default=FailureStrategy.RETRY, description="Failure handling strategy"
57
+ )
58
+ retry_config: RetryConfig = Field(
59
+ default_factory=RetryConfig, description="Retry configuration"
60
+ )
61
+ condition: str | None = Field(None, description="Conditional execution expression")
62
+
63
+
64
+ class RetrievalLimits(BaseModel):
65
+ """Operator policy for Knowledge Store retrieval inside a stage.
66
+
67
+ Caps only — these shrink what an analyzer asks for, never expand it.
68
+ The orchestrator passes this on the per-analysis request payload;
69
+ ``KnowledgeClient.for_analysis(email)`` enforces them transparently.
70
+ """
71
+
72
+ max_top_k: int | None = Field(
73
+ default=None,
74
+ ge=1,
75
+ le=1000,
76
+ description="Cap on top_k requested by any analyzer in this stage",
77
+ )
78
+ min_score_floor: float | None = Field(
79
+ default=None,
80
+ ge=0.0,
81
+ le=1.0,
82
+ description="Raise the min_score floor even if analyzer asked for less",
83
+ )
84
+
85
+
86
+ class WorkflowStage(BaseModel):
87
+ """A stage in a workflow with multiple analyzers."""
88
+
89
+ name: str = Field(..., description="Stage name")
90
+ type: str = Field(
91
+ default="sequential",
92
+ description="Stage type (parallel, sequential, conditional)",
93
+ )
94
+ analyzers: list[AnalyzerConfig] = Field(
95
+ default_factory=list, description="List of analyzers in this stage"
96
+ )
97
+ retrieval_limits: RetrievalLimits | None = Field(
98
+ default=None,
99
+ description="Operator-policy caps on knowledge-store retrieval for this stage",
100
+ )
101
+ depends_on: list[str] = Field(
102
+ default_factory=list, description="Names of stages this depends on"
103
+ )
104
+ continue_on_failure: bool = Field(
105
+ default=True, description="Continue if some analyzers fail"
106
+ )
107
+ min_successful_analyzers: int = Field(
108
+ default=1, ge=0, description="Minimum successful analyzers required"
109
+ )
110
+ aggregation_method: str = Field(
111
+ default="weighted_average", description="Method for aggregating results"
112
+ )
113
+ condition: str | None = Field(None, description="Conditional execution expression")
114
+
115
+
116
+ class Workflow(BaseModel):
117
+ """Complete workflow configuration for email analysis."""
118
+
119
+ id: str = Field(..., description="Workflow ID")
120
+ name: str = Field(..., description="Workflow name")
121
+ version: int = Field(default=1, ge=1, description="Workflow version")
122
+ description: str = Field(default="", description="Workflow description")
123
+ stages: list[WorkflowStage] = Field(
124
+ default_factory=list, description="List of workflow stages"
125
+ )
126
+ final_stage_name: str = Field(
127
+ default="",
128
+ description="Name of the final decision stage. Empty means aggregate all results.",
129
+ )
130
+ timeout_ms: int = Field(
131
+ default=300000,
132
+ ge=1000,
133
+ le=600000,
134
+ description="Total workflow timeout in milliseconds",
135
+ )
136
+ max_parallel_analyzers: int = Field(
137
+ default=10, ge=1, le=100, description="Maximum parallel analyzer executions"
138
+ )
139
+ created_by: str = Field(default="system", description="Creator of the workflow")
@@ -0,0 +1,353 @@
1
+ Metadata-Version: 2.4
2
+ Name: spot-sdk-python
3
+ Version: 1.0.0
4
+ Summary: Python SDK for SPOT platform - API contracts, models, and utilities
5
+ License: Apache-2.0
6
+ Author: SPOT Project
7
+ Author-email: spot@sonn.lu
8
+ Requires-Python: >=3.11,<4.0
9
+ Classifier: License :: OSI Approved :: Apache Software License
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
13
+ Classifier: Programming Language :: Python :: 3.13
14
+ Classifier: Programming Language :: Python :: 3.14
15
+ Requires-Dist: httpx (>=0.25.0,<0.26.0)
16
+ Requires-Dist: pydantic-settings (>=2.0.0,<3.0.0)
17
+ Requires-Dist: pydantic[email] (>=2.0.0,<3.0.0)
18
+ Requires-Dist: typing-extensions (>=4.8.0,<5.0.0)
19
+ Project-URL: Homepage, https://sonn.lu
20
+ Description-Content-Type: text/markdown
21
+
22
+ # SPOT Contracts Python SDK
23
+
24
+ Python SDK generated from SPOT Platform OpenAPI specifications.
25
+
26
+ ## Installation
27
+
28
+ ```bash
29
+ pip install spot-sdk-python
30
+ ```
31
+
32
+ ## Usage
33
+
34
+ ### Basic Usage
35
+
36
+ ```python
37
+ from spot_sdk.analyzer import Email, EmailHeader, AnalysisResult
38
+ from spot_sdk.api_gateway import AnalysisRequest
39
+
40
+ # Create email header
41
+ header = EmailHeader(
42
+ message_id="msg-001",
43
+ subject="Test Email",
44
+ sender="sender@example.com",
45
+ recipients=["recipient@example.com"],
46
+ date="2024-01-01T12:00:00Z"
47
+ )
48
+
49
+ # Create email object
50
+ email = Email(
51
+ id="12345",
52
+ headers=header,
53
+ body_text="This is the email content"
54
+ )
55
+
56
+ # Use with API Gateway
57
+ request = AnalysisRequest(email=email.dict())
58
+ ```
59
+
60
+ ### Pydantic Models
61
+
62
+ All models are Pydantic v2 BaseModel instances with:
63
+
64
+ - **Type validation** - Automatic validation of field types
65
+ - **Serialization** - `.dict()` and `.json()` methods
66
+ - **Field descriptions** - Documentation from OpenAPI specs
67
+ - **IDE support** - Full type hints and autocompletion
68
+
69
+ ### Example: Email Analysis
70
+
71
+ ```python
72
+ import httpx
73
+ from spot_sdk.analyzer import Email
74
+ from spot_sdk.api_gateway import AnalysisRequest
75
+
76
+ async def analyze_email(email: Email) -> dict:
77
+ async with httpx.AsyncClient() as client:
78
+ request = AnalysisRequest(email=email.dict())
79
+ response = await client.post(
80
+ "http://localhost:8001/api/v1/analyze",
81
+ json=request.dict()
82
+ )
83
+ return response.json()
84
+
85
+ # Usage
86
+ email = Email(
87
+ id="sample-123",
88
+ headers={
89
+ "message_id": "msg-sample",
90
+ "subject": "Urgent: Action Required",
91
+ "sender": "suspicious@example.com",
92
+ "recipients": ["target@company.com"],
93
+ "date": "2024-01-15T10:30:00Z"
94
+ },
95
+ body_text="Please click this link immediately..."
96
+ )
97
+
98
+ result = await analyze_email(email)
99
+ print(f"Phishing detected: {result['is_phishing']}")
100
+ ```
101
+
102
+ ### Accessing Previous Stage Results (`analysis_context`)
103
+
104
+ The orchestrator automatically populates `Email.analysis_context` with the results of all previously-completed stages before calling each analyzer. No workflow configuration is required -- analyzers can read whatever they need.
105
+
106
+ **Structure:**
107
+
108
+ ```python
109
+ email.analysis_context = {
110
+ "<stage-name>": {
111
+ "providers": {
112
+ "<provider-id>": { ...free-form data... }
113
+ },
114
+ "analyzers": {
115
+ "<analyzer-id>": { ...AnalyzerResult fields... }
116
+ }
117
+ },
118
+ ...
119
+ }
120
+ ```
121
+
122
+ - Top-level keys are stage names
123
+ - Each stage has a `providers` dict (populated by Context Providers) and an `analyzers` dict
124
+ - Analyzer values expose all `AnalyzerResult` fields (`is_phishing`, `confidence`, `threat_level`, `indicators`, `analyzer_details`, ...)
125
+ - Only stages that have completed before the current analyzer runs are present
126
+
127
+ **Recommended: use the `email.ctx` helper**
128
+
129
+ The SDK provides an `AnalysisContextReader` (via the `Email.ctx` property) with ergonomic accessors so you don't have to write manual dict traversal:
130
+
131
+ ```python
132
+ @app.post("/internal/analyze")
133
+ async def analyze_email(email: Email) -> AnalysisResult:
134
+ ctx = email.ctx # AnalysisContextReader
135
+
136
+ # Get a specific analyzer from any stage (first match)
137
+ nlp = ctx.analyzer("analyzer-nlp")
138
+ if nlp:
139
+ confidence = nlp["confidence"]
140
+
141
+ # Get a specific analyzer from a specific stage
142
+ ml = ctx.analyzer("analyzer-ml", stage="parallel-analysis")
143
+
144
+ # Get a provider's data
145
+ dept = ctx.provider("employee-dir")
146
+ if dept:
147
+ role = dept.get("role")
148
+
149
+ # Iterate all analyzers in a stage
150
+ for aid, result in ctx.analyzers_in("parallel-analysis").items():
151
+ print(aid, result["confidence"])
152
+
153
+ # Listing helpers
154
+ ctx.stages() # ['enrichment', 'parallel-analysis']
155
+ ctx.analyzer_ids() # ['analyzer-nlp', 'analyzer-ml']
156
+ ctx.provider_ids() # ['employee-dir', 'threat-feed']
157
+ ctx.analyzer_ids_in("parallel-analysis") # stage-scoped
158
+ ctx.provider_ids_in("enrichment")
159
+ ctx.has_stage("parallel-analysis") # True/False
160
+
161
+ # Bulk accessors: {stage: {id: data, ...}, ...}
162
+ ctx.all_analyzers()
163
+ ctx.all_providers()
164
+ ...
165
+ ```
166
+
167
+ All single accessors (`analyzer()`, `provider()`) return `None` when not found. Bulk accessors return empty dicts. Nothing raises -- missing stages or IDs are treated as normal.
168
+
169
+ **Raw access (if you prefer)**
170
+
171
+ You can also read `email.analysis_context` directly as a plain dict:
172
+
173
+ ```python
174
+ prev = email.analysis_context.get("parallel-analysis", {})
175
+ nlp = prev.get("analyzers", {}).get("analyzer-nlp")
176
+ ```
177
+
178
+ Analyzers that don't need previous results can safely ignore this field -- it defaults to an empty dict.
179
+
180
+ ## Plugins
181
+
182
+ SPOT's umbrella vocabulary for pluggable components is **plugin**.
183
+ ``PluginKind`` discriminates between the two kinds today:
184
+
185
+ ```python
186
+ from spot_sdk import PluginKind
187
+
188
+ PluginKind.ANALYZER # analyzer kind
189
+ PluginKind.CONTEXT_PROVIDER # context provider kind
190
+ ```
191
+
192
+ - **Analyzers** run ``POST /internal/analyze`` and return an
193
+ ``AnalysisResult`` (a phishing verdict contributing to aggregation).
194
+ - **Context providers** run ``POST /internal/enrich`` and return an
195
+ ``EnrichmentResult`` whose data populates ``analysis_context`` for
196
+ downstream analyzers.
197
+
198
+ Plugins are discovered from OCI image labels prefixed ``spot.plugin.*``
199
+ (see the platform documentation for the label contract) and installed
200
+ into the platform via the ``/api/v1/plugins/*`` and
201
+ ``/api/v1/config/plugin/{kind}`` APIs.
202
+
203
+ ## Context Providers
204
+
205
+ Context providers enrich an email with organizational data (employee directory,
206
+ sender history, threat intelligence, knowledge bases, ...) before analyzers run.
207
+ They are defined per workflow stage and run in parallel, just like analyzers.
208
+
209
+ Each provider exposes a single HTTP endpoint:
210
+
211
+ ```
212
+ POST /internal/enrich
213
+ Content-Type: application/json
214
+
215
+ Input: Email (same model analyzers receive)
216
+ Output: EnrichmentResult
217
+ ```
218
+
219
+ ### Implementing a provider
220
+
221
+ ```python
222
+ from fastapi import FastAPI
223
+ from spot_sdk import Email, EnrichmentResult
224
+
225
+ app = FastAPI()
226
+
227
+ @app.post("/internal/enrich")
228
+ async def enrich(email: Email) -> EnrichmentResult:
229
+ sender = email.headers.sender
230
+ return EnrichmentResult(
231
+ provider_id="employee-dir",
232
+ data={
233
+ "sender_known": True,
234
+ "sender_department": "Finance",
235
+ "sender_role": "CFO",
236
+ },
237
+ source="company-ldap",
238
+ confidence=0.95,
239
+ )
240
+ ```
241
+
242
+ The orchestrator merges each provider's `data` dict into
243
+ `email.analysis_context["<stage-name>"]["providers"]["<provider-id>"]`,
244
+ where downstream analyzers can read it via `email.ctx.provider(...)`.
245
+
246
+ ## Knowledge Store
247
+
248
+ The Knowledge Store is the platform's RAG layer: context providers
249
+ deposit tagged documents, analyzers fetch them on demand. Neither side
250
+ references the other — they share only the tag vocabulary.
251
+
252
+ ```python
253
+ from spot_sdk import KnowledgeClient, KnowledgeDocument, KnowledgeTag, chunk_text
254
+ ```
255
+
256
+ ### Ingestion (providers)
257
+
258
+ Upsert is idempotent on `id`. Use stable deterministic ids so re-syncs
259
+ replace rather than duplicate.
260
+
261
+ ```python
262
+ kb = KnowledgeClient(
263
+ url=os.environ["SPOT_KNOWLEDGE_URL"], # injected by installer
264
+ api_key=os.environ["SPOT_INTERNAL_API_KEY"], # idem
265
+ )
266
+
267
+ await kb.bulk_upsert([
268
+ KnowledgeDocument(
269
+ id=f"employee:{e.email}",
270
+ content=f"{e.name}, {e.title}, {e.department}",
271
+ tags=[KnowledgeTag.EMPLOYEE, *([KnowledgeTag.EXECUTIVE] if e.is_exec else [])],
272
+ metadata={"email": e.email, "title": e.title},
273
+ source="provider-employee-dir",
274
+ )
275
+ for e in employees
276
+ ])
277
+ ```
278
+
279
+ For long content use `chunk_text(body, max_chars=2000, overlap=200)`
280
+ and upsert each chunk with `metadata["parent_id"]` set to the source
281
+ doc id.
282
+
283
+ ### Consumption (analyzers)
284
+
285
+ ```python
286
+ kb = KnowledgeClient.for_analysis(email) # reads SPOT_KNOWLEDGE_URL +
287
+ # email.retrieval_limits
288
+ docs = await kb.fetch(tags="employee+executive", text=email.sender, top_k=3)
289
+ ```
290
+
291
+ Tag-expression syntax: `a`, `a+b` (AND), `a|b` (OR), `a+b|c`
292
+ (`(a AND b) OR c`). `+` binds tighter than `|`. Empty or `None` means
293
+ no tag filter.
294
+
295
+ Workflow-declared `retrieval_limits` (stage-level caps on `top_k` and
296
+ `min_score`) are enforced transparently by `for_analysis()` — the
297
+ analyzer doesn't need to know about them.
298
+
299
+ ### Testing
300
+
301
+ ```python
302
+ from spot_sdk.testing.fake_knowledge_client import FakeKnowledgeClient
303
+
304
+ kb = FakeKnowledgeClient()
305
+ await kb.upsert(KnowledgeDocument(id="e:a", content="Alice", tags=["employee"]))
306
+ assert (await kb.fetch(tags="employee", text="Alice"))[0].id == "e:a"
307
+ ```
308
+
309
+ ## Available Models
310
+
311
+ ### Analyzer Service (`spot_sdk.analyzer`)
312
+ - `Email` - Email data structure
313
+ - `EmailHeader` - Email header fields
314
+ - `Attachment` - Email attachment
315
+ - `AnalysisResult` - Analysis results
316
+ - `AnalysisIndicator` - Phishing indicators
317
+ - `ThreatLevel` - Threat level enum
318
+
319
+ ### API Gateway (`spot_sdk.api_gateway`)
320
+ - `AnalysisRequest` - Analysis request
321
+ - `AnalysisResponse` - Analysis response
322
+ - `ConfigRequest` - Configuration request
323
+ - `ConfigResponse` - Configuration response
324
+
325
+ ### Workflow (`spot_sdk.workflow`)
326
+ - `Workflow`, `WorkflowStage`
327
+ - `AnalyzerConfig`, `ContextProviderConfig`
328
+ - `RetryConfig`, `FailureStrategy`
329
+
330
+ ### Plugin vocabulary (`spot_sdk.plugin`)
331
+ - `PluginKind` - `ANALYZER` | `CONTEXT_PROVIDER`
332
+
333
+ ### Enrichment (`spot_sdk.enrichment`)
334
+ - `EnrichmentResult` - Returned by context providers from `/internal/enrich`
335
+
336
+ ## Requirements
337
+
338
+ - Python 3.11+
339
+ - pydantic >= 2.0.0
340
+ - httpx >= 0.25.0
341
+ - typing-extensions >= 4.8.0
342
+
343
+ ## Development
344
+
345
+ This SDK is automatically generated from OpenAPI specifications.
346
+ Do not modify generated files directly - update the OpenAPI specs instead.
347
+
348
+ ## Version
349
+
350
+ Current version: 1.0.0
351
+
352
+ Generated from SPOT Contracts repository.
353
+
@@ -0,0 +1,27 @@
1
+ spot_sdk/__init__.py,sha256=upOtpYDG0mhdii-ZC5tCY0qoYoMEG3-i-ZxLYoilmEA,2289
2
+ spot_sdk/analysis_context.py,sha256=ghdGhmWUOrkCrrEZWNn9nbVinMW1kmtei2xOKZyVGBI,3605
3
+ spot_sdk/analyzer.py,sha256=bDrXVOO5YlP6M9Vh390XczWVTsgUaJlbR2ziibQwsag,2594
4
+ spot_sdk/analyzer_base.py,sha256=ADrVNfyxtZWRff9UCgN0IGzFWkCUioRHwOKw7MAbtHQ,10223
5
+ spot_sdk/api_gateway.py,sha256=8p_CPqP77gQPqgiBQzgfHhco-j1GnoucOMEQKx9bM4A,6222
6
+ spot_sdk/config.py,sha256=7nkYo4gJ4tWLAqoA_dNAzbmasPmSxw8WesyDJ9tm_IA,1878
7
+ spot_sdk/config_client.py,sha256=7L9ws9UPYLgfZ-QsLl1MBlHXrRxjmi90Oy37vSnSfu8,6515
8
+ spot_sdk/config_helpers.py,sha256=_L72GjQ339SUQhGweAZQLzeqCYr-NwfT-F2aLweRiDo,839
9
+ spot_sdk/email.py,sha256=uL7DIFOHCUN0aKTVb6zv8hmQN16GpJFQ2aBUqTfHdbI,4399
10
+ spot_sdk/errors.py,sha256=HTaRaoDfn5XV4v03qD4eCM-IgNaR7upQ08UF-I-1qaI,813
11
+ spot_sdk/knowledge.py,sha256=b1wNUrkFpqHK3nD65fT3n4adFA2cgeFhyoKs7E9gJmo,11930
12
+ spot_sdk/knowledge_tags.py,sha256=BGhDsNeF8Q0MMNEIZjWrgQWjwH_Vf7dGBNN_hlBqTXs,978
13
+ spot_sdk/logging.py,sha256=lY-OdTvShp3Vb-1Gmhi88z810YGFhh6y21PbNscBGSg,3907
14
+ spot_sdk/ollama.py,sha256=Vaa1r0-boy0Gz6pxQqXplUMIdpLh_H1l75uIo9xRbO0,1794
15
+ spot_sdk/orchestrator.py,sha256=GwZkvHyyONEmJVDChJCb8OT98GJ-vQTng2PHhD39Aso,2985
16
+ spot_sdk/plugin.py,sha256=JGib8UoS9vDH7rK7-aAmnCab4CH1t2Zk1qy7O-Oz4AU,981
17
+ spot_sdk/results.py,sha256=IdPHksHJ6zTY4zXS2tp4jk_N--Yxnf49Edswc1G-3vY,4226
18
+ spot_sdk/settings_schema.py,sha256=FKoIWTxPaLDara2LMaXtLudJ9Xa_-OR2meC4m63iycE,1916
19
+ spot_sdk/testing/README.md,sha256=EIRGybgaY8sQ6J6U5poezhz4HyK_pmFrANPMnWCEQSM,1782
20
+ spot_sdk/testing/__init__.py,sha256=G8F0FNePLMcKn-pocQQMzA_iv1WcbdJFfwBWW7JqlJ4,483
21
+ spot_sdk/testing/factories.py,sha256=ADJL0JkzS3DsEc4m9T0seuzPJuQuF7aWJaBUbIdecgY,4917
22
+ spot_sdk/testing/fake_knowledge_client.py,sha256=aGLYqFHtHTqJvN0j1VDRmaLIgEEbpQyxuAC4XGJUuK4,3729
23
+ spot_sdk/threat_levels.py,sha256=eRkd2GBS0TCKw4VGERkFYGMa-fxhrxB7TMa2UHi7e_Q,910
24
+ spot_sdk/workflow.py,sha256=CYtAIuTJYN1qS0rmZzQXBrAgOYb46v8aCHJP7qzEHdU,4586
25
+ spot_sdk_python-1.0.0.dist-info/METADATA,sha256=9OTIeSzpBxzwC4MJbsWbn2F1PsHt89qwabWMCBupK9k,10682
26
+ spot_sdk_python-1.0.0.dist-info/WHEEL,sha256=Vz2fHgx6HFtSwhs8KvkHLqH5Ea4w1_rner5uNVGCeIE,88
27
+ spot_sdk_python-1.0.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: poetry-core 2.3.2
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any