spot-sdk-python 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,271 @@
1
+ # generated by datamodel-codegen:
2
+ # filename: api-gateway.yaml
3
+
4
+ from __future__ import annotations
5
+
6
+ from enum import StrEnum
7
+ from typing import Any
8
+
9
+ from pydantic import AwareDatetime, BaseModel, EmailStr, Field, SecretStr, constr
10
+
11
+
12
+ class Features(BaseModel):
13
+ configuration_api: bool | None = None
14
+ database_config: bool | None = None
15
+ oauth: bool | None = None
16
+ rate_limiting: bool | None = None
17
+ audit_logging: bool | None = None
18
+
19
+
20
+ class HealthStatus(BaseModel):
21
+ status: str | None = Field(None, examples=['healthy'])
22
+ service: str | None = Field(None, examples=['api-gateway'])
23
+ version: str | None = None
24
+ features: Features | None = None
25
+
26
+
27
+ class Token(BaseModel):
28
+ access_token: str
29
+ refresh_token: str
30
+ token_type: str = Field(..., examples=['bearer'])
31
+ expires_in: int
32
+
33
+
34
+ class RefreshRequest(BaseModel):
35
+ refresh_token: str
36
+ """
37
+ Valid refresh token
38
+ """
39
+
40
+
41
+ class UserRole(StrEnum):
42
+ admin = 'admin'
43
+ analyst = 'analyst'
44
+ viewer = 'viewer'
45
+
46
+
47
+ class UserCreate(BaseModel):
48
+ username: constr(min_length=3, max_length=50)
49
+ email: EmailStr
50
+ password: SecretStr
51
+ role: UserRole | None = 'analyst'
52
+
53
+
54
+ class UserUpdate(BaseModel):
55
+ username: constr(min_length=3, max_length=50) | None = None
56
+ email: EmailStr | None = None
57
+
58
+
59
+ class UserRoleUpdate(BaseModel):
60
+ role: UserRole
61
+
62
+
63
+ class UserPasswordUpdate(BaseModel):
64
+ current_password: str | None = None
65
+ new_password: constr(min_length=8)
66
+
67
+
68
+ class UserProfile(BaseModel):
69
+ id: str | None = None
70
+ username: str | None = None
71
+ email: str | None = None
72
+ role: UserRole | None = None
73
+ is_active: bool | None = None
74
+ created_at: str | None = None
75
+ updated_at: str | None = None
76
+
77
+
78
+ class UserListResponse(BaseModel):
79
+ users: list[UserProfile]
80
+ total: int
81
+ limit: int
82
+ offset: int
83
+
84
+
85
+ class MessageResponse(BaseModel):
86
+ message: str | None = None
87
+
88
+
89
+ class AnalyzerConfigResponse(BaseModel):
90
+ analyzer_id: str | None = None
91
+ enabled: bool | None = None
92
+ settings: dict[str, Any] | None = None
93
+ config_version: str | None = None
94
+
95
+
96
+ class WorkflowsConfigResponse(BaseModel):
97
+ workflows: list[dict[str, Any]] | None = None
98
+ config_version: str | None = None
99
+
100
+
101
+ class ConfigReloadResponse(BaseModel):
102
+ old_version: str | None = None
103
+ new_version: str | None = None
104
+ changed: dict[str, Any] | None = None
105
+
106
+
107
+ class PluginInfo(BaseModel):
108
+ name: str | None = None
109
+ type: str | None = None
110
+ version: str | None = None
111
+ enabled: bool | None = None
112
+ created_at: AwareDatetime | None = None
113
+
114
+
115
+ class PluginListResponse(BaseModel):
116
+ plugins: list[PluginInfo] | None = None
117
+ total: int | None = None
118
+
119
+
120
+ class Type(StrEnum):
121
+ oci = 'oci'
122
+ git = 'git'
123
+ bundle = 'bundle'
124
+
125
+
126
+ class Source(BaseModel):
127
+ type: Type
128
+ ref: str | None = None
129
+
130
+
131
+ class PluginRequireRequest(BaseModel):
132
+ source: Source
133
+ dry_run: bool | None = False
134
+
135
+
136
+ class AnalysisRequest(BaseModel):
137
+ email: dict[str, Any]
138
+ analyzers: list[str] | None = None
139
+ priority: str | None = 'normal'
140
+ workflow_id: str | None = None
141
+
142
+
143
+ class AnalysisResponse(BaseModel):
144
+ job_id: str | None = None
145
+ email_id: str | None = None
146
+ status: str | None = None
147
+ workflow: str | None = None
148
+ message: str | None = None
149
+
150
+
151
+ class EmailAnalysisStatus(BaseModel):
152
+ job_id: str | None = None
153
+ email_id: str | None = None
154
+ workflow_id: str | None = None
155
+ """
156
+ ID of the workflow used for analysis
157
+ """
158
+ status: str | None = None
159
+ created_at: AwareDatetime | None = None
160
+ started_at: AwareDatetime | None = None
161
+ completed_at: AwareDatetime | None = None
162
+ result: dict[str, Any] | None = None
163
+ error: str | None = None
164
+
165
+
166
+ class AnalysisStatsResponse(BaseModel):
167
+ total: int | None = None
168
+ """
169
+ Total number of analysis jobs
170
+ """
171
+ by_status: dict[str, int] | None = None
172
+ """
173
+ Count of jobs by status
174
+ """
175
+ by_threat_level: dict[str, int] | None = None
176
+ """
177
+ Count of completed jobs by threat level
178
+ """
179
+ phishing_count: int | None = None
180
+ """
181
+ Number of emails classified as phishing
182
+ """
183
+ legitimate_count: int | None = None
184
+ """
185
+ Number of emails classified as legitimate
186
+ """
187
+ avg_confidence: float | None = None
188
+ """
189
+ Average confidence score (0.0-1.0)
190
+ """
191
+
192
+
193
+ class EmailResponse(BaseModel):
194
+ id: str
195
+ original_id: str | None = None
196
+ subject: str | None = None
197
+ body_text: str | None = None
198
+ body_html: str | None = None
199
+ headers: dict[str, Any] | None = None
200
+ attachments: list[dict[str, Any]] | None = None
201
+ language: str | None = None
202
+ source: str | None = None
203
+ received_at: str | None = None
204
+ created_at: str | None = None
205
+ analyzed_at: str | None = None
206
+ expires_at: str | None = None
207
+
208
+
209
+ class EmailListResponse(BaseModel):
210
+ emails: list[dict[str, Any]]
211
+ total: int
212
+ limit: int
213
+ offset: int
214
+
215
+
216
+ class EmailDeleteResponse(BaseModel):
217
+ deleted: bool
218
+ message: str
219
+
220
+
221
+ class EmailSourcesResponse(BaseModel):
222
+ sources: list[str]
223
+
224
+
225
+ class WorkflowSummary(BaseModel):
226
+ id: str | None = None
227
+ name: str | None = None
228
+ version: int | None = None
229
+ description: str | None = None
230
+ stages: int | None = None
231
+ final_stage: str | None = None
232
+
233
+
234
+ class WorkflowTestRequest(BaseModel):
235
+ email: dict[str, Any]
236
+ wait_for_completion: bool | None = True
237
+
238
+
239
+ class RetrieverInfo(BaseModel):
240
+ id: str | None = None
241
+ url: str | None = None
242
+ priority: int | None = None
243
+ enabled: bool | None = None
244
+ capabilities: list[str] | None = None
245
+ status: str | None = None
246
+ last_check: str | None = None
247
+
248
+
249
+ class Source1(BaseModel):
250
+ id: str
251
+ type: str
252
+ config: dict[str, Any]
253
+ enabled: bool | None = True
254
+ fetch_interval: int | None = 300
255
+
256
+
257
+ class FetchRequest(BaseModel):
258
+ source: Source1
259
+ priority: str | None = 'normal'
260
+
261
+
262
+ class FetchJobStatus(BaseModel):
263
+ job_id: str | None = None
264
+ source_id: str | None = None
265
+ retriever_id: str | None = None
266
+ status: str | None = None
267
+ created_at: str | None = None
268
+ started_at: str | None = None
269
+ completed_at: str | None = None
270
+ emails_fetched: int | None = 0
271
+ error: str | None = None
spot_sdk/config.py ADDED
@@ -0,0 +1,46 @@
1
+ """Configuration management models for SPOT platform."""
2
+
3
+ from typing import Any
4
+
5
+ from pydantic import BaseModel, Field
6
+
7
+
8
+ class ConfigOption(BaseModel):
9
+ """Single configuration option with full metadata."""
10
+
11
+ path: str = Field(description="Dot-notation path (e.g., 'platform.log_level')")
12
+ value: Any = Field(description="Current value")
13
+ default: Any = Field(description="Default value from schema")
14
+ type: str = Field(description="JSON Schema type (string, boolean, integer, etc.)")
15
+ description: str = Field(default="", description="Human-readable description")
16
+ is_default: bool = Field(description="True if current value equals default")
17
+ enum: list[str] | None = Field(
18
+ default=None, description="Allowed values if restricted"
19
+ )
20
+ minimum: float | None = Field(default=None, description="Minimum value for numbers")
21
+ maximum: float | None = Field(default=None, description="Maximum value for numbers")
22
+
23
+
24
+ class ConfigStatus(BaseModel):
25
+ """Configuration synchronization status."""
26
+
27
+ applied_hash: str = Field(description="Hash of last-applied configuration")
28
+ pending_hash: str = Field(description="Hash of current in-memory configuration")
29
+ needs_reload: bool = Field(description="True if pending differs from applied")
30
+ pending_changes: int = Field(
31
+ default=0, description="Number of options changed since last reload"
32
+ )
33
+
34
+
35
+ class ConfigUpdateRequest(BaseModel):
36
+ """Request to update a single configuration option."""
37
+
38
+ value: Any = Field(description="New value for the option")
39
+
40
+
41
+ class ConfigReloadResult(BaseModel):
42
+ """Result of a configuration reload operation."""
43
+
44
+ status: ConfigStatus = Field(description="Updated configuration status")
45
+ reloaded: bool = Field(description="True if reload was performed")
46
+ message: str = Field(default="", description="Optional status message")
@@ -0,0 +1,203 @@
1
+ """Configuration client for SPOT analyzers.
2
+
3
+ Provides:
4
+ - Remote config fetching from core platform
5
+ - Graceful handling of network errors (preserves last known good config)
6
+ - Simple interface for getting config values
7
+
8
+ Error handling:
9
+ - On 404: Reset config (no central config, use local defaults)
10
+ - On network/timeout errors: Keep last known good config unchanged
11
+
12
+ Usage in analyzer main.py:
13
+ from spot_sdk.config_client import ConfigClient
14
+
15
+ config_client = ConfigClient(
16
+ analyzer_id="spot-analyzer-nlp",
17
+ platform_url="http://api-gateway:8000",
18
+ )
19
+
20
+ # Fetch config on startup
21
+ await config_client.fetch_config()
22
+
23
+ # Get config values (with fallback to defaults)
24
+ threshold = config_client.get("confidence_threshold", 0.7)
25
+
26
+ # Reload config (called from /admin/reload-config endpoint)
27
+ await config_client.reload()
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ from typing import Any
33
+
34
+ import httpx
35
+
36
+ from .logging import get_logger
37
+
38
+ logger = get_logger(__name__)
39
+
40
+
41
+ class ConfigClient:
42
+ """Client for fetching configuration from SPOT platform.
43
+
44
+ Analyzers use this to fetch their centralized configuration overrides
45
+ from the api-gateway's /api/v1/config/analyzer/{id} endpoint.
46
+
47
+ The config is stored internally and can be accessed via get() method.
48
+ On network errors, the last known good config is preserved.
49
+ On 404, config is reset (meaning no central config exists).
50
+ """
51
+
52
+ def __init__(
53
+ self,
54
+ analyzer_id: str,
55
+ platform_url: str = "http://api-gateway:8000",
56
+ timeout: float = 10.0,
57
+ ) -> None:
58
+ """Initialize ConfigClient.
59
+
60
+ Args:
61
+ analyzer_id: The analyzer identifier (e.g., "spot-analyzer-nlp")
62
+ platform_url: Base URL of the api-gateway (default: http://api-gateway:8000)
63
+ timeout: HTTP request timeout in seconds (default: 10.0)
64
+ """
65
+ self.analyzer_id = analyzer_id
66
+ self.platform_url = platform_url.rstrip("/")
67
+ self.timeout = timeout
68
+
69
+ # Internal state
70
+ self._config: dict[str, Any] = {}
71
+ self._version: str | None = None
72
+ self._enabled: bool = True
73
+
74
+ @property
75
+ def config(self) -> dict[str, Any]:
76
+ """Get the current configuration dict."""
77
+ return self._config
78
+
79
+ @property
80
+ def version(self) -> str | None:
81
+ """Get the current config version (None if never fetched successfully)."""
82
+ return self._version
83
+
84
+ @property
85
+ def enabled(self) -> bool:
86
+ """Check if analyzer is enabled in central config."""
87
+ return self._enabled
88
+
89
+ async def fetch_config(self) -> dict[str, Any]:
90
+ """Fetch configuration from platform.
91
+
92
+ Makes HTTP GET request to /api/v1/config/analyzer/{analyzer_id}.
93
+
94
+ On success: Updates internal state with new settings.
95
+ On 404: Resets config to empty (no central config, use local defaults).
96
+ On network error: Keeps last known good config unchanged.
97
+
98
+ Returns:
99
+ Dict of configuration settings (may be unchanged on network error).
100
+ """
101
+ url = f"{self.platform_url}/api/v1/config/analyzer/{self.analyzer_id}"
102
+
103
+ try:
104
+ async with httpx.AsyncClient(timeout=self.timeout) as client:
105
+ response = await client.get(url)
106
+
107
+ if response.status_code == 404:
108
+ # No central config for this analyzer - reset to empty
109
+ logger.info(
110
+ "No central config for %s, using local defaults",
111
+ self.analyzer_id,
112
+ )
113
+ self._config = {}
114
+ self._version = None
115
+ self._enabled = True
116
+ return self._config
117
+
118
+ response.raise_for_status()
119
+ data = response.json()
120
+
121
+ # Update state only on successful response
122
+ self._config = data.get("settings", {})
123
+ self._version = data.get("config_version")
124
+ self._enabled = data.get("enabled", True)
125
+
126
+ logger.info(
127
+ "Fetched config for %s (version=%s, enabled=%s, keys=%s)",
128
+ self.analyzer_id,
129
+ self._version,
130
+ self._enabled,
131
+ list(self._config.keys()),
132
+ )
133
+
134
+ return self._config
135
+
136
+ except httpx.ConnectError as e:
137
+ # Network error - keep last known good config
138
+ logger.warning(
139
+ "Cannot connect to platform at %s: %s. Keeping current config.",
140
+ self.platform_url,
141
+ e,
142
+ )
143
+ return self._config
144
+
145
+ except httpx.TimeoutException:
146
+ # Timeout - keep last known good config
147
+ logger.warning(
148
+ "Timeout fetching config from %s. Keeping current config.",
149
+ url,
150
+ )
151
+ return self._config
152
+
153
+ except httpx.HTTPStatusError as e:
154
+ # HTTP error (not 404) - keep last known good config
155
+ logger.warning(
156
+ "HTTP error fetching config for %s: %s. Keeping current config.",
157
+ self.analyzer_id,
158
+ e,
159
+ )
160
+ return self._config
161
+
162
+ except Exception as e:
163
+ # Unexpected error - keep last known good config
164
+ logger.warning(
165
+ "Unexpected error fetching config for %s: %s. Keeping current config.",
166
+ self.analyzer_id,
167
+ e,
168
+ )
169
+ return self._config
170
+
171
+ async def reload(self) -> dict[str, Any]:
172
+ """Reload configuration from platform.
173
+
174
+ Alias for fetch_config(), called when analyzer receives reload signal.
175
+
176
+ Returns:
177
+ Dict of configuration settings.
178
+ """
179
+ logger.info("Reloading config for %s", self.analyzer_id)
180
+ return await self.fetch_config()
181
+
182
+ def get(self, key: str, default: Any = None) -> Any:
183
+ """Get a configuration value.
184
+
185
+ Args:
186
+ key: Configuration key name
187
+ default: Default value if key not found
188
+
189
+ Returns:
190
+ Configuration value or default.
191
+ """
192
+ return self._config.get(key, default)
193
+
194
+ def has(self, key: str) -> bool:
195
+ """Check if a configuration key exists.
196
+
197
+ Args:
198
+ key: Configuration key name
199
+
200
+ Returns:
201
+ True if key exists in config.
202
+ """
203
+ return key in self._config
@@ -0,0 +1,25 @@
1
+ """Shared configuration helpers for SPOT analyzers."""
2
+
3
+ from typing import Any, TypeVar
4
+
5
+ from pydantic_settings import BaseSettings
6
+
7
+ T = TypeVar("T", bound=BaseSettings)
8
+
9
+
10
+ def merge_settings(settings_class: type[T], base: T, overrides: dict[str, Any]) -> T:
11
+ """Merge base settings with central overrides.
12
+
13
+ Only applies overrides for fields that exist in the Settings model.
14
+
15
+ Args:
16
+ settings_class: The Pydantic Settings class (used for field discovery).
17
+ base: Base Settings instance from local env/defaults.
18
+ overrides: Override values from central config.
19
+
20
+ Returns:
21
+ New Settings instance with overrides applied.
22
+ """
23
+ valid_keys = set(settings_class.model_fields.keys())
24
+ filtered = {k: v for k, v in overrides.items() if k in valid_keys}
25
+ return base.model_copy(update=filtered)
spot_sdk/email.py ADDED
@@ -0,0 +1,136 @@
1
+ """Email data models."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import uuid
6
+ from datetime import datetime
7
+ from typing import TYPE_CHECKING, Any
8
+
9
+ from pydantic import BaseModel, EmailStr, Field, field_validator, model_validator
10
+
11
+ if TYPE_CHECKING:
12
+ from .analysis_context import AnalysisContextReader
13
+
14
+
15
+ class EmailHeader(BaseModel):
16
+ """Email header fields."""
17
+
18
+ message_id: str
19
+ subject: str
20
+ sender: EmailStr
21
+ recipients: list[EmailStr] = Field(..., min_length=1)
22
+ cc: list[EmailStr] = Field(default_factory=list)
23
+ bcc: list[EmailStr] = Field(default_factory=list)
24
+ date: datetime
25
+ reply_to: EmailStr | None = None
26
+ return_path: EmailStr | None = None
27
+
28
+ # Raw headers for advanced analysis
29
+ raw_headers: dict[str, str] = Field(default_factory=dict)
30
+
31
+ @field_validator("recipients", "cc", "bcc", mode="before")
32
+ @classmethod
33
+ def ensure_list(cls, v: Any) -> list[str]:
34
+ if isinstance(v, str):
35
+ return [v]
36
+ return v or []
37
+
38
+
39
+ class Attachment(BaseModel):
40
+ """Email attachment."""
41
+
42
+ filename: str
43
+ content_type: str
44
+ size_bytes: int
45
+ content_hash: str # SHA256 hash
46
+ is_inline: bool = False
47
+
48
+ # Optional: actual content for analysis (base64 encoded)
49
+ content: str | None = None
50
+
51
+ @field_validator("content_hash")
52
+ @classmethod
53
+ def validate_hash(cls, v: str) -> str:
54
+ if len(v) != 64: # SHA256 is 64 hex chars
55
+ raise ValueError("content_hash must be a valid SHA256 hash")
56
+ return v
57
+
58
+
59
+ class Email(BaseModel):
60
+ """
61
+ Complete email representation for analysis.
62
+
63
+ This model contains all email data needed for phishing detection
64
+ while maintaining GDPR compliance through optional anonymization.
65
+ """
66
+
67
+ id: str = Field(default_factory=lambda: str(uuid.uuid4()))
68
+ headers: EmailHeader
69
+ body_text: str | None = None
70
+ body_html: str | None = None
71
+ attachments: list[Attachment] = Field(default_factory=list)
72
+
73
+ # Metadata
74
+ received_at: datetime = Field(default_factory=datetime.utcnow)
75
+ language: str | None = None # ISO 639-1 language code
76
+ source: str = "unknown" # imap, smtp, api, etc.
77
+
78
+ # Analysis context
79
+ organization_context: dict[str, str] = Field(default_factory=dict)
80
+
81
+ # Context from previously-executed workflow stages, populated by the orchestrator.
82
+ #
83
+ # Structure:
84
+ # {
85
+ # "<stage-name>": {
86
+ # "providers": {"<provider-id>": {...data...}},
87
+ # "analyzers": {"<analyzer-id>": {...AnalyzerResult fields...}}
88
+ # },
89
+ # ...
90
+ # }
91
+ #
92
+ # Only stages that have already completed are included. Empty dict by default.
93
+ analysis_context: dict[str, Any] = Field(default_factory=dict)
94
+
95
+ # Operator policy for Knowledge Store retrieval inside the current stage.
96
+ # Populated by the orchestrator from WorkflowStage.retrieval_limits and
97
+ # consumed transparently by KnowledgeClient.for_analysis(email). Empty
98
+ # dict means "no caps" (use analyzer-supplied parameters as-is).
99
+ retrieval_limits: dict[str, Any] = Field(default_factory=dict)
100
+
101
+ @model_validator(mode="after")
102
+ def ensure_content(self) -> "Email":
103
+ """Ensure at least one body format is provided."""
104
+ if not self.body_text and not self.body_html:
105
+ raise ValueError("Email must have either body_text or body_html")
106
+ return self
107
+
108
+ @property
109
+ def content(self) -> str:
110
+ """Get email content, preferring text over HTML."""
111
+ return self.body_text or self.body_html or ""
112
+
113
+ @property
114
+ def has_attachments(self) -> bool:
115
+ """Check if email has attachments."""
116
+ return len(self.attachments) > 0
117
+
118
+ @property
119
+ def ctx(self) -> "AnalysisContextReader":
120
+ """Ergonomic accessor for analysis_context from previous stages.
121
+
122
+ Example:
123
+ nlp = email.ctx.analyzer("analyzer-nlp")
124
+ if nlp:
125
+ confidence = nlp["confidence"]
126
+ """
127
+ # Imported lazily to avoid a circular reference at module import time
128
+ from .analysis_context import AnalysisContextReader
129
+
130
+ return AnalysisContextReader(self.analysis_context)
131
+
132
+ def get_urls(self) -> list[str]:
133
+ """Extract URLs from email content (to be implemented in analyzer)."""
134
+ # This would be implemented in the actual analyzer
135
+ # Placeholder for interface definition
136
+ return []
spot_sdk/errors.py ADDED
@@ -0,0 +1,24 @@
1
+ """Standard error response models for SPOT API."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from pydantic import BaseModel, Field
8
+
9
+
10
+ class ErrorResponse(BaseModel):
11
+ """Standard error response for all SPOT API endpoints.
12
+
13
+ Provides a consistent error format across the platform:
14
+ - ``error``: machine-readable error code (e.g. "not_found", "validation_error")
15
+ - ``message``: human-readable explanation
16
+ - ``details``: optional structured context (validation errors, field info, etc.)
17
+ """
18
+
19
+ error: str = Field(..., description="Machine-readable error code")
20
+ message: str = Field(..., description="Human-readable error message")
21
+ details: dict[str, Any] = Field(
22
+ default_factory=dict,
23
+ description="Optional structured details about the error",
24
+ )