spot-sdk-python 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spot_sdk/__init__.py +86 -0
- spot_sdk/analysis_context.py +99 -0
- spot_sdk/analyzer.py +106 -0
- spot_sdk/analyzer_base.py +316 -0
- spot_sdk/api_gateway.py +271 -0
- spot_sdk/config.py +46 -0
- spot_sdk/config_client.py +203 -0
- spot_sdk/config_helpers.py +25 -0
- spot_sdk/email.py +136 -0
- spot_sdk/errors.py +24 -0
- spot_sdk/knowledge.py +341 -0
- spot_sdk/knowledge_tags.py +31 -0
- spot_sdk/logging.py +133 -0
- spot_sdk/ollama.py +58 -0
- spot_sdk/orchestrator.py +70 -0
- spot_sdk/plugin.py +30 -0
- spot_sdk/results.py +129 -0
- spot_sdk/settings_schema.py +56 -0
- spot_sdk/testing/README.md +83 -0
- spot_sdk/testing/__init__.py +22 -0
- spot_sdk/testing/factories.py +177 -0
- spot_sdk/testing/fake_knowledge_client.py +105 -0
- spot_sdk/threat_levels.py +33 -0
- spot_sdk/workflow.py +139 -0
- spot_sdk_python-1.0.0.dist-info/METADATA +353 -0
- spot_sdk_python-1.0.0.dist-info/RECORD +27 -0
- spot_sdk_python-1.0.0.dist-info/WHEEL +4 -0
spot_sdk/api_gateway.py
ADDED
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
# generated by datamodel-codegen:
|
|
2
|
+
# filename: api-gateway.yaml
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
from enum import StrEnum
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from pydantic import AwareDatetime, BaseModel, EmailStr, Field, SecretStr, constr
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Features(BaseModel):
|
|
13
|
+
configuration_api: bool | None = None
|
|
14
|
+
database_config: bool | None = None
|
|
15
|
+
oauth: bool | None = None
|
|
16
|
+
rate_limiting: bool | None = None
|
|
17
|
+
audit_logging: bool | None = None
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class HealthStatus(BaseModel):
|
|
21
|
+
status: str | None = Field(None, examples=['healthy'])
|
|
22
|
+
service: str | None = Field(None, examples=['api-gateway'])
|
|
23
|
+
version: str | None = None
|
|
24
|
+
features: Features | None = None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class Token(BaseModel):
|
|
28
|
+
access_token: str
|
|
29
|
+
refresh_token: str
|
|
30
|
+
token_type: str = Field(..., examples=['bearer'])
|
|
31
|
+
expires_in: int
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class RefreshRequest(BaseModel):
|
|
35
|
+
refresh_token: str
|
|
36
|
+
"""
|
|
37
|
+
Valid refresh token
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class UserRole(StrEnum):
|
|
42
|
+
admin = 'admin'
|
|
43
|
+
analyst = 'analyst'
|
|
44
|
+
viewer = 'viewer'
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class UserCreate(BaseModel):
|
|
48
|
+
username: constr(min_length=3, max_length=50)
|
|
49
|
+
email: EmailStr
|
|
50
|
+
password: SecretStr
|
|
51
|
+
role: UserRole | None = 'analyst'
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class UserUpdate(BaseModel):
|
|
55
|
+
username: constr(min_length=3, max_length=50) | None = None
|
|
56
|
+
email: EmailStr | None = None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class UserRoleUpdate(BaseModel):
|
|
60
|
+
role: UserRole
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class UserPasswordUpdate(BaseModel):
|
|
64
|
+
current_password: str | None = None
|
|
65
|
+
new_password: constr(min_length=8)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class UserProfile(BaseModel):
|
|
69
|
+
id: str | None = None
|
|
70
|
+
username: str | None = None
|
|
71
|
+
email: str | None = None
|
|
72
|
+
role: UserRole | None = None
|
|
73
|
+
is_active: bool | None = None
|
|
74
|
+
created_at: str | None = None
|
|
75
|
+
updated_at: str | None = None
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class UserListResponse(BaseModel):
|
|
79
|
+
users: list[UserProfile]
|
|
80
|
+
total: int
|
|
81
|
+
limit: int
|
|
82
|
+
offset: int
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class MessageResponse(BaseModel):
|
|
86
|
+
message: str | None = None
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class AnalyzerConfigResponse(BaseModel):
|
|
90
|
+
analyzer_id: str | None = None
|
|
91
|
+
enabled: bool | None = None
|
|
92
|
+
settings: dict[str, Any] | None = None
|
|
93
|
+
config_version: str | None = None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class WorkflowsConfigResponse(BaseModel):
|
|
97
|
+
workflows: list[dict[str, Any]] | None = None
|
|
98
|
+
config_version: str | None = None
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class ConfigReloadResponse(BaseModel):
|
|
102
|
+
old_version: str | None = None
|
|
103
|
+
new_version: str | None = None
|
|
104
|
+
changed: dict[str, Any] | None = None
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
class PluginInfo(BaseModel):
|
|
108
|
+
name: str | None = None
|
|
109
|
+
type: str | None = None
|
|
110
|
+
version: str | None = None
|
|
111
|
+
enabled: bool | None = None
|
|
112
|
+
created_at: AwareDatetime | None = None
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
class PluginListResponse(BaseModel):
|
|
116
|
+
plugins: list[PluginInfo] | None = None
|
|
117
|
+
total: int | None = None
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class Type(StrEnum):
|
|
121
|
+
oci = 'oci'
|
|
122
|
+
git = 'git'
|
|
123
|
+
bundle = 'bundle'
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
class Source(BaseModel):
|
|
127
|
+
type: Type
|
|
128
|
+
ref: str | None = None
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class PluginRequireRequest(BaseModel):
|
|
132
|
+
source: Source
|
|
133
|
+
dry_run: bool | None = False
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class AnalysisRequest(BaseModel):
|
|
137
|
+
email: dict[str, Any]
|
|
138
|
+
analyzers: list[str] | None = None
|
|
139
|
+
priority: str | None = 'normal'
|
|
140
|
+
workflow_id: str | None = None
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
class AnalysisResponse(BaseModel):
|
|
144
|
+
job_id: str | None = None
|
|
145
|
+
email_id: str | None = None
|
|
146
|
+
status: str | None = None
|
|
147
|
+
workflow: str | None = None
|
|
148
|
+
message: str | None = None
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class EmailAnalysisStatus(BaseModel):
|
|
152
|
+
job_id: str | None = None
|
|
153
|
+
email_id: str | None = None
|
|
154
|
+
workflow_id: str | None = None
|
|
155
|
+
"""
|
|
156
|
+
ID of the workflow used for analysis
|
|
157
|
+
"""
|
|
158
|
+
status: str | None = None
|
|
159
|
+
created_at: AwareDatetime | None = None
|
|
160
|
+
started_at: AwareDatetime | None = None
|
|
161
|
+
completed_at: AwareDatetime | None = None
|
|
162
|
+
result: dict[str, Any] | None = None
|
|
163
|
+
error: str | None = None
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
class AnalysisStatsResponse(BaseModel):
|
|
167
|
+
total: int | None = None
|
|
168
|
+
"""
|
|
169
|
+
Total number of analysis jobs
|
|
170
|
+
"""
|
|
171
|
+
by_status: dict[str, int] | None = None
|
|
172
|
+
"""
|
|
173
|
+
Count of jobs by status
|
|
174
|
+
"""
|
|
175
|
+
by_threat_level: dict[str, int] | None = None
|
|
176
|
+
"""
|
|
177
|
+
Count of completed jobs by threat level
|
|
178
|
+
"""
|
|
179
|
+
phishing_count: int | None = None
|
|
180
|
+
"""
|
|
181
|
+
Number of emails classified as phishing
|
|
182
|
+
"""
|
|
183
|
+
legitimate_count: int | None = None
|
|
184
|
+
"""
|
|
185
|
+
Number of emails classified as legitimate
|
|
186
|
+
"""
|
|
187
|
+
avg_confidence: float | None = None
|
|
188
|
+
"""
|
|
189
|
+
Average confidence score (0.0-1.0)
|
|
190
|
+
"""
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
class EmailResponse(BaseModel):
|
|
194
|
+
id: str
|
|
195
|
+
original_id: str | None = None
|
|
196
|
+
subject: str | None = None
|
|
197
|
+
body_text: str | None = None
|
|
198
|
+
body_html: str | None = None
|
|
199
|
+
headers: dict[str, Any] | None = None
|
|
200
|
+
attachments: list[dict[str, Any]] | None = None
|
|
201
|
+
language: str | None = None
|
|
202
|
+
source: str | None = None
|
|
203
|
+
received_at: str | None = None
|
|
204
|
+
created_at: str | None = None
|
|
205
|
+
analyzed_at: str | None = None
|
|
206
|
+
expires_at: str | None = None
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
class EmailListResponse(BaseModel):
|
|
210
|
+
emails: list[dict[str, Any]]
|
|
211
|
+
total: int
|
|
212
|
+
limit: int
|
|
213
|
+
offset: int
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
class EmailDeleteResponse(BaseModel):
|
|
217
|
+
deleted: bool
|
|
218
|
+
message: str
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
class EmailSourcesResponse(BaseModel):
|
|
222
|
+
sources: list[str]
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
class WorkflowSummary(BaseModel):
|
|
226
|
+
id: str | None = None
|
|
227
|
+
name: str | None = None
|
|
228
|
+
version: int | None = None
|
|
229
|
+
description: str | None = None
|
|
230
|
+
stages: int | None = None
|
|
231
|
+
final_stage: str | None = None
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
class WorkflowTestRequest(BaseModel):
|
|
235
|
+
email: dict[str, Any]
|
|
236
|
+
wait_for_completion: bool | None = True
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
class RetrieverInfo(BaseModel):
|
|
240
|
+
id: str | None = None
|
|
241
|
+
url: str | None = None
|
|
242
|
+
priority: int | None = None
|
|
243
|
+
enabled: bool | None = None
|
|
244
|
+
capabilities: list[str] | None = None
|
|
245
|
+
status: str | None = None
|
|
246
|
+
last_check: str | None = None
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
class Source1(BaseModel):
|
|
250
|
+
id: str
|
|
251
|
+
type: str
|
|
252
|
+
config: dict[str, Any]
|
|
253
|
+
enabled: bool | None = True
|
|
254
|
+
fetch_interval: int | None = 300
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
class FetchRequest(BaseModel):
|
|
258
|
+
source: Source1
|
|
259
|
+
priority: str | None = 'normal'
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
class FetchJobStatus(BaseModel):
|
|
263
|
+
job_id: str | None = None
|
|
264
|
+
source_id: str | None = None
|
|
265
|
+
retriever_id: str | None = None
|
|
266
|
+
status: str | None = None
|
|
267
|
+
created_at: str | None = None
|
|
268
|
+
started_at: str | None = None
|
|
269
|
+
completed_at: str | None = None
|
|
270
|
+
emails_fetched: int | None = 0
|
|
271
|
+
error: str | None = None
|
spot_sdk/config.py
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Configuration management models for SPOT platform."""
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from pydantic import BaseModel, Field
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class ConfigOption(BaseModel):
|
|
9
|
+
"""Single configuration option with full metadata."""
|
|
10
|
+
|
|
11
|
+
path: str = Field(description="Dot-notation path (e.g., 'platform.log_level')")
|
|
12
|
+
value: Any = Field(description="Current value")
|
|
13
|
+
default: Any = Field(description="Default value from schema")
|
|
14
|
+
type: str = Field(description="JSON Schema type (string, boolean, integer, etc.)")
|
|
15
|
+
description: str = Field(default="", description="Human-readable description")
|
|
16
|
+
is_default: bool = Field(description="True if current value equals default")
|
|
17
|
+
enum: list[str] | None = Field(
|
|
18
|
+
default=None, description="Allowed values if restricted"
|
|
19
|
+
)
|
|
20
|
+
minimum: float | None = Field(default=None, description="Minimum value for numbers")
|
|
21
|
+
maximum: float | None = Field(default=None, description="Maximum value for numbers")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ConfigStatus(BaseModel):
|
|
25
|
+
"""Configuration synchronization status."""
|
|
26
|
+
|
|
27
|
+
applied_hash: str = Field(description="Hash of last-applied configuration")
|
|
28
|
+
pending_hash: str = Field(description="Hash of current in-memory configuration")
|
|
29
|
+
needs_reload: bool = Field(description="True if pending differs from applied")
|
|
30
|
+
pending_changes: int = Field(
|
|
31
|
+
default=0, description="Number of options changed since last reload"
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class ConfigUpdateRequest(BaseModel):
|
|
36
|
+
"""Request to update a single configuration option."""
|
|
37
|
+
|
|
38
|
+
value: Any = Field(description="New value for the option")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class ConfigReloadResult(BaseModel):
|
|
42
|
+
"""Result of a configuration reload operation."""
|
|
43
|
+
|
|
44
|
+
status: ConfigStatus = Field(description="Updated configuration status")
|
|
45
|
+
reloaded: bool = Field(description="True if reload was performed")
|
|
46
|
+
message: str = Field(default="", description="Optional status message")
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
"""Configuration client for SPOT analyzers.
|
|
2
|
+
|
|
3
|
+
Provides:
|
|
4
|
+
- Remote config fetching from core platform
|
|
5
|
+
- Graceful handling of network errors (preserves last known good config)
|
|
6
|
+
- Simple interface for getting config values
|
|
7
|
+
|
|
8
|
+
Error handling:
|
|
9
|
+
- On 404: Reset config (no central config, use local defaults)
|
|
10
|
+
- On network/timeout errors: Keep last known good config unchanged
|
|
11
|
+
|
|
12
|
+
Usage in analyzer main.py:
|
|
13
|
+
from spot_sdk.config_client import ConfigClient
|
|
14
|
+
|
|
15
|
+
config_client = ConfigClient(
|
|
16
|
+
analyzer_id="spot-analyzer-nlp",
|
|
17
|
+
platform_url="http://api-gateway:8000",
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
# Fetch config on startup
|
|
21
|
+
await config_client.fetch_config()
|
|
22
|
+
|
|
23
|
+
# Get config values (with fallback to defaults)
|
|
24
|
+
threshold = config_client.get("confidence_threshold", 0.7)
|
|
25
|
+
|
|
26
|
+
# Reload config (called from /admin/reload-config endpoint)
|
|
27
|
+
await config_client.reload()
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
from typing import Any
|
|
33
|
+
|
|
34
|
+
import httpx
|
|
35
|
+
|
|
36
|
+
from .logging import get_logger
|
|
37
|
+
|
|
38
|
+
logger = get_logger(__name__)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class ConfigClient:
|
|
42
|
+
"""Client for fetching configuration from SPOT platform.
|
|
43
|
+
|
|
44
|
+
Analyzers use this to fetch their centralized configuration overrides
|
|
45
|
+
from the api-gateway's /api/v1/config/analyzer/{id} endpoint.
|
|
46
|
+
|
|
47
|
+
The config is stored internally and can be accessed via get() method.
|
|
48
|
+
On network errors, the last known good config is preserved.
|
|
49
|
+
On 404, config is reset (meaning no central config exists).
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
analyzer_id: str,
|
|
55
|
+
platform_url: str = "http://api-gateway:8000",
|
|
56
|
+
timeout: float = 10.0,
|
|
57
|
+
) -> None:
|
|
58
|
+
"""Initialize ConfigClient.
|
|
59
|
+
|
|
60
|
+
Args:
|
|
61
|
+
analyzer_id: The analyzer identifier (e.g., "spot-analyzer-nlp")
|
|
62
|
+
platform_url: Base URL of the api-gateway (default: http://api-gateway:8000)
|
|
63
|
+
timeout: HTTP request timeout in seconds (default: 10.0)
|
|
64
|
+
"""
|
|
65
|
+
self.analyzer_id = analyzer_id
|
|
66
|
+
self.platform_url = platform_url.rstrip("/")
|
|
67
|
+
self.timeout = timeout
|
|
68
|
+
|
|
69
|
+
# Internal state
|
|
70
|
+
self._config: dict[str, Any] = {}
|
|
71
|
+
self._version: str | None = None
|
|
72
|
+
self._enabled: bool = True
|
|
73
|
+
|
|
74
|
+
@property
|
|
75
|
+
def config(self) -> dict[str, Any]:
|
|
76
|
+
"""Get the current configuration dict."""
|
|
77
|
+
return self._config
|
|
78
|
+
|
|
79
|
+
@property
|
|
80
|
+
def version(self) -> str | None:
|
|
81
|
+
"""Get the current config version (None if never fetched successfully)."""
|
|
82
|
+
return self._version
|
|
83
|
+
|
|
84
|
+
@property
|
|
85
|
+
def enabled(self) -> bool:
|
|
86
|
+
"""Check if analyzer is enabled in central config."""
|
|
87
|
+
return self._enabled
|
|
88
|
+
|
|
89
|
+
async def fetch_config(self) -> dict[str, Any]:
|
|
90
|
+
"""Fetch configuration from platform.
|
|
91
|
+
|
|
92
|
+
Makes HTTP GET request to /api/v1/config/analyzer/{analyzer_id}.
|
|
93
|
+
|
|
94
|
+
On success: Updates internal state with new settings.
|
|
95
|
+
On 404: Resets config to empty (no central config, use local defaults).
|
|
96
|
+
On network error: Keeps last known good config unchanged.
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
Dict of configuration settings (may be unchanged on network error).
|
|
100
|
+
"""
|
|
101
|
+
url = f"{self.platform_url}/api/v1/config/analyzer/{self.analyzer_id}"
|
|
102
|
+
|
|
103
|
+
try:
|
|
104
|
+
async with httpx.AsyncClient(timeout=self.timeout) as client:
|
|
105
|
+
response = await client.get(url)
|
|
106
|
+
|
|
107
|
+
if response.status_code == 404:
|
|
108
|
+
# No central config for this analyzer - reset to empty
|
|
109
|
+
logger.info(
|
|
110
|
+
"No central config for %s, using local defaults",
|
|
111
|
+
self.analyzer_id,
|
|
112
|
+
)
|
|
113
|
+
self._config = {}
|
|
114
|
+
self._version = None
|
|
115
|
+
self._enabled = True
|
|
116
|
+
return self._config
|
|
117
|
+
|
|
118
|
+
response.raise_for_status()
|
|
119
|
+
data = response.json()
|
|
120
|
+
|
|
121
|
+
# Update state only on successful response
|
|
122
|
+
self._config = data.get("settings", {})
|
|
123
|
+
self._version = data.get("config_version")
|
|
124
|
+
self._enabled = data.get("enabled", True)
|
|
125
|
+
|
|
126
|
+
logger.info(
|
|
127
|
+
"Fetched config for %s (version=%s, enabled=%s, keys=%s)",
|
|
128
|
+
self.analyzer_id,
|
|
129
|
+
self._version,
|
|
130
|
+
self._enabled,
|
|
131
|
+
list(self._config.keys()),
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
return self._config
|
|
135
|
+
|
|
136
|
+
except httpx.ConnectError as e:
|
|
137
|
+
# Network error - keep last known good config
|
|
138
|
+
logger.warning(
|
|
139
|
+
"Cannot connect to platform at %s: %s. Keeping current config.",
|
|
140
|
+
self.platform_url,
|
|
141
|
+
e,
|
|
142
|
+
)
|
|
143
|
+
return self._config
|
|
144
|
+
|
|
145
|
+
except httpx.TimeoutException:
|
|
146
|
+
# Timeout - keep last known good config
|
|
147
|
+
logger.warning(
|
|
148
|
+
"Timeout fetching config from %s. Keeping current config.",
|
|
149
|
+
url,
|
|
150
|
+
)
|
|
151
|
+
return self._config
|
|
152
|
+
|
|
153
|
+
except httpx.HTTPStatusError as e:
|
|
154
|
+
# HTTP error (not 404) - keep last known good config
|
|
155
|
+
logger.warning(
|
|
156
|
+
"HTTP error fetching config for %s: %s. Keeping current config.",
|
|
157
|
+
self.analyzer_id,
|
|
158
|
+
e,
|
|
159
|
+
)
|
|
160
|
+
return self._config
|
|
161
|
+
|
|
162
|
+
except Exception as e:
|
|
163
|
+
# Unexpected error - keep last known good config
|
|
164
|
+
logger.warning(
|
|
165
|
+
"Unexpected error fetching config for %s: %s. Keeping current config.",
|
|
166
|
+
self.analyzer_id,
|
|
167
|
+
e,
|
|
168
|
+
)
|
|
169
|
+
return self._config
|
|
170
|
+
|
|
171
|
+
async def reload(self) -> dict[str, Any]:
|
|
172
|
+
"""Reload configuration from platform.
|
|
173
|
+
|
|
174
|
+
Alias for fetch_config(), called when analyzer receives reload signal.
|
|
175
|
+
|
|
176
|
+
Returns:
|
|
177
|
+
Dict of configuration settings.
|
|
178
|
+
"""
|
|
179
|
+
logger.info("Reloading config for %s", self.analyzer_id)
|
|
180
|
+
return await self.fetch_config()
|
|
181
|
+
|
|
182
|
+
def get(self, key: str, default: Any = None) -> Any:
|
|
183
|
+
"""Get a configuration value.
|
|
184
|
+
|
|
185
|
+
Args:
|
|
186
|
+
key: Configuration key name
|
|
187
|
+
default: Default value if key not found
|
|
188
|
+
|
|
189
|
+
Returns:
|
|
190
|
+
Configuration value or default.
|
|
191
|
+
"""
|
|
192
|
+
return self._config.get(key, default)
|
|
193
|
+
|
|
194
|
+
def has(self, key: str) -> bool:
|
|
195
|
+
"""Check if a configuration key exists.
|
|
196
|
+
|
|
197
|
+
Args:
|
|
198
|
+
key: Configuration key name
|
|
199
|
+
|
|
200
|
+
Returns:
|
|
201
|
+
True if key exists in config.
|
|
202
|
+
"""
|
|
203
|
+
return key in self._config
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Shared configuration helpers for SPOT analyzers."""
|
|
2
|
+
|
|
3
|
+
from typing import Any, TypeVar
|
|
4
|
+
|
|
5
|
+
from pydantic_settings import BaseSettings
|
|
6
|
+
|
|
7
|
+
T = TypeVar("T", bound=BaseSettings)
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def merge_settings(settings_class: type[T], base: T, overrides: dict[str, Any]) -> T:
|
|
11
|
+
"""Merge base settings with central overrides.
|
|
12
|
+
|
|
13
|
+
Only applies overrides for fields that exist in the Settings model.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
settings_class: The Pydantic Settings class (used for field discovery).
|
|
17
|
+
base: Base Settings instance from local env/defaults.
|
|
18
|
+
overrides: Override values from central config.
|
|
19
|
+
|
|
20
|
+
Returns:
|
|
21
|
+
New Settings instance with overrides applied.
|
|
22
|
+
"""
|
|
23
|
+
valid_keys = set(settings_class.model_fields.keys())
|
|
24
|
+
filtered = {k: v for k, v in overrides.items() if k in valid_keys}
|
|
25
|
+
return base.model_copy(update=filtered)
|
spot_sdk/email.py
ADDED
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""Email data models."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import uuid
|
|
6
|
+
from datetime import datetime
|
|
7
|
+
from typing import TYPE_CHECKING, Any
|
|
8
|
+
|
|
9
|
+
from pydantic import BaseModel, EmailStr, Field, field_validator, model_validator
|
|
10
|
+
|
|
11
|
+
if TYPE_CHECKING:
|
|
12
|
+
from .analysis_context import AnalysisContextReader
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class EmailHeader(BaseModel):
|
|
16
|
+
"""Email header fields."""
|
|
17
|
+
|
|
18
|
+
message_id: str
|
|
19
|
+
subject: str
|
|
20
|
+
sender: EmailStr
|
|
21
|
+
recipients: list[EmailStr] = Field(..., min_length=1)
|
|
22
|
+
cc: list[EmailStr] = Field(default_factory=list)
|
|
23
|
+
bcc: list[EmailStr] = Field(default_factory=list)
|
|
24
|
+
date: datetime
|
|
25
|
+
reply_to: EmailStr | None = None
|
|
26
|
+
return_path: EmailStr | None = None
|
|
27
|
+
|
|
28
|
+
# Raw headers for advanced analysis
|
|
29
|
+
raw_headers: dict[str, str] = Field(default_factory=dict)
|
|
30
|
+
|
|
31
|
+
@field_validator("recipients", "cc", "bcc", mode="before")
|
|
32
|
+
@classmethod
|
|
33
|
+
def ensure_list(cls, v: Any) -> list[str]:
|
|
34
|
+
if isinstance(v, str):
|
|
35
|
+
return [v]
|
|
36
|
+
return v or []
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Attachment(BaseModel):
|
|
40
|
+
"""Email attachment."""
|
|
41
|
+
|
|
42
|
+
filename: str
|
|
43
|
+
content_type: str
|
|
44
|
+
size_bytes: int
|
|
45
|
+
content_hash: str # SHA256 hash
|
|
46
|
+
is_inline: bool = False
|
|
47
|
+
|
|
48
|
+
# Optional: actual content for analysis (base64 encoded)
|
|
49
|
+
content: str | None = None
|
|
50
|
+
|
|
51
|
+
@field_validator("content_hash")
|
|
52
|
+
@classmethod
|
|
53
|
+
def validate_hash(cls, v: str) -> str:
|
|
54
|
+
if len(v) != 64: # SHA256 is 64 hex chars
|
|
55
|
+
raise ValueError("content_hash must be a valid SHA256 hash")
|
|
56
|
+
return v
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Email(BaseModel):
|
|
60
|
+
"""
|
|
61
|
+
Complete email representation for analysis.
|
|
62
|
+
|
|
63
|
+
This model contains all email data needed for phishing detection
|
|
64
|
+
while maintaining GDPR compliance through optional anonymization.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
id: str = Field(default_factory=lambda: str(uuid.uuid4()))
|
|
68
|
+
headers: EmailHeader
|
|
69
|
+
body_text: str | None = None
|
|
70
|
+
body_html: str | None = None
|
|
71
|
+
attachments: list[Attachment] = Field(default_factory=list)
|
|
72
|
+
|
|
73
|
+
# Metadata
|
|
74
|
+
received_at: datetime = Field(default_factory=datetime.utcnow)
|
|
75
|
+
language: str | None = None # ISO 639-1 language code
|
|
76
|
+
source: str = "unknown" # imap, smtp, api, etc.
|
|
77
|
+
|
|
78
|
+
# Analysis context
|
|
79
|
+
organization_context: dict[str, str] = Field(default_factory=dict)
|
|
80
|
+
|
|
81
|
+
# Context from previously-executed workflow stages, populated by the orchestrator.
|
|
82
|
+
#
|
|
83
|
+
# Structure:
|
|
84
|
+
# {
|
|
85
|
+
# "<stage-name>": {
|
|
86
|
+
# "providers": {"<provider-id>": {...data...}},
|
|
87
|
+
# "analyzers": {"<analyzer-id>": {...AnalyzerResult fields...}}
|
|
88
|
+
# },
|
|
89
|
+
# ...
|
|
90
|
+
# }
|
|
91
|
+
#
|
|
92
|
+
# Only stages that have already completed are included. Empty dict by default.
|
|
93
|
+
analysis_context: dict[str, Any] = Field(default_factory=dict)
|
|
94
|
+
|
|
95
|
+
# Operator policy for Knowledge Store retrieval inside the current stage.
|
|
96
|
+
# Populated by the orchestrator from WorkflowStage.retrieval_limits and
|
|
97
|
+
# consumed transparently by KnowledgeClient.for_analysis(email). Empty
|
|
98
|
+
# dict means "no caps" (use analyzer-supplied parameters as-is).
|
|
99
|
+
retrieval_limits: dict[str, Any] = Field(default_factory=dict)
|
|
100
|
+
|
|
101
|
+
@model_validator(mode="after")
|
|
102
|
+
def ensure_content(self) -> "Email":
|
|
103
|
+
"""Ensure at least one body format is provided."""
|
|
104
|
+
if not self.body_text and not self.body_html:
|
|
105
|
+
raise ValueError("Email must have either body_text or body_html")
|
|
106
|
+
return self
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def content(self) -> str:
|
|
110
|
+
"""Get email content, preferring text over HTML."""
|
|
111
|
+
return self.body_text or self.body_html or ""
|
|
112
|
+
|
|
113
|
+
@property
|
|
114
|
+
def has_attachments(self) -> bool:
|
|
115
|
+
"""Check if email has attachments."""
|
|
116
|
+
return len(self.attachments) > 0
|
|
117
|
+
|
|
118
|
+
@property
|
|
119
|
+
def ctx(self) -> "AnalysisContextReader":
|
|
120
|
+
"""Ergonomic accessor for analysis_context from previous stages.
|
|
121
|
+
|
|
122
|
+
Example:
|
|
123
|
+
nlp = email.ctx.analyzer("analyzer-nlp")
|
|
124
|
+
if nlp:
|
|
125
|
+
confidence = nlp["confidence"]
|
|
126
|
+
"""
|
|
127
|
+
# Imported lazily to avoid a circular reference at module import time
|
|
128
|
+
from .analysis_context import AnalysisContextReader
|
|
129
|
+
|
|
130
|
+
return AnalysisContextReader(self.analysis_context)
|
|
131
|
+
|
|
132
|
+
def get_urls(self) -> list[str]:
|
|
133
|
+
"""Extract URLs from email content (to be implemented in analyzer)."""
|
|
134
|
+
# This would be implemented in the actual analyzer
|
|
135
|
+
# Placeholder for interface definition
|
|
136
|
+
return []
|
spot_sdk/errors.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Standard error response models for SPOT API."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel, Field
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ErrorResponse(BaseModel):
|
|
11
|
+
"""Standard error response for all SPOT API endpoints.
|
|
12
|
+
|
|
13
|
+
Provides a consistent error format across the platform:
|
|
14
|
+
- ``error``: machine-readable error code (e.g. "not_found", "validation_error")
|
|
15
|
+
- ``message``: human-readable explanation
|
|
16
|
+
- ``details``: optional structured context (validation errors, field info, etc.)
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
error: str = Field(..., description="Machine-readable error code")
|
|
20
|
+
message: str = Field(..., description="Human-readable error message")
|
|
21
|
+
details: dict[str, Any] = Field(
|
|
22
|
+
default_factory=dict,
|
|
23
|
+
description="Optional structured details about the error",
|
|
24
|
+
)
|