flowsense-engine 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- flowsense/__init__.py +77 -0
- flowsense/application/__init__.py +20 -0
- flowsense/application/analyzer.py +68 -0
- flowsense/application/output.py +128 -0
- flowsense/application/pipeline.py +210 -0
- flowsense/application/ports.py +11 -0
- flowsense/application/request.py +32 -0
- flowsense/application/serialization.py +177 -0
- flowsense/cli/__init__.py +0 -0
- flowsense/cli/main.py +195 -0
- flowsense/cli/report.py +233 -0
- flowsense/collector/__init__.py +0 -0
- flowsense/collector/airflow_client.py +5 -0
- flowsense/config.py +72 -0
- flowsense/domain/__init__.py +52 -0
- flowsense/domain/enums.py +47 -0
- flowsense/domain/exceptions.py +25 -0
- flowsense/domain/models.py +19 -0
- flowsense/domain/policy.py +55 -0
- flowsense/domain/results.py +187 -0
- flowsense/engine/__init__.py +0 -0
- flowsense/engine/analyzer.py +17 -0
- flowsense/engine/change_point.py +87 -0
- flowsense/engine/drift.py +80 -0
- flowsense/engine/history.py +34 -0
- flowsense/engine/impact.py +51 -0
- flowsense/engine/propagation.py +173 -0
- flowsense/engine/root_cause.py +148 -0
- flowsense/engine/timing.py +221 -0
- flowsense/engine/trend.py +82 -0
- flowsense/infrastructure/__init__.py +1 -0
- flowsense/infrastructure/airflow/__init__.py +13 -0
- flowsense/infrastructure/airflow/client.py +373 -0
- flowsense/infrastructure/airflow/dto.py +33 -0
- flowsense/infrastructure/airflow/exceptions.py +31 -0
- flowsense/infrastructure/airflow/mapper.py +27 -0
- flowsense/mcp/__init__.py +0 -0
- flowsense/mcp/server.py +75 -0
- flowsense/models/__init__.py +7 -0
- flowsense/models/dag_analysis.py +3 -0
- flowsense/models/task_run.py +3 -0
- flowsense/py.typed +0 -0
- flowsense/version.py +6 -0
- flowsense_engine-0.2.1.dist-info/METADATA +447 -0
- flowsense_engine-0.2.1.dist-info/RECORD +49 -0
- flowsense_engine-0.2.1.dist-info/WHEEL +5 -0
- flowsense_engine-0.2.1.dist-info/entry_points.txt +3 -0
- flowsense_engine-0.2.1.dist-info/licenses/LICENSE +17 -0
- flowsense_engine-0.2.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
from flowsense.domain.enums import (
|
|
2
|
+
ChangeDirection,
|
|
3
|
+
ImpactClassification,
|
|
4
|
+
MappedTaskAggregation,
|
|
5
|
+
Severity,
|
|
6
|
+
TrendDirection,
|
|
7
|
+
severity_meets_threshold,
|
|
8
|
+
)
|
|
9
|
+
from flowsense.domain.exceptions import (
|
|
10
|
+
ConfigurationError,
|
|
11
|
+
FlowSenseError,
|
|
12
|
+
InsufficientHistoryError,
|
|
13
|
+
InvalidTaskTimingError,
|
|
14
|
+
)
|
|
15
|
+
from flowsense.domain.models import TaskRun
|
|
16
|
+
from flowsense.domain.policy import DEFAULT_ANALYSIS_POLICY, AnalysisPolicy
|
|
17
|
+
from flowsense.domain.results import (
|
|
18
|
+
AnalysisDiagnostic,
|
|
19
|
+
ChangePointResult,
|
|
20
|
+
DAGAnalysis,
|
|
21
|
+
DAGAnalysisSummary,
|
|
22
|
+
DriftResult,
|
|
23
|
+
PropagationResult,
|
|
24
|
+
RootCauseResult,
|
|
25
|
+
TaskImpact,
|
|
26
|
+
TrendResult,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
__all__ = [
|
|
30
|
+
"DEFAULT_ANALYSIS_POLICY",
|
|
31
|
+
"AnalysisDiagnostic",
|
|
32
|
+
"AnalysisPolicy",
|
|
33
|
+
"ChangeDirection",
|
|
34
|
+
"ChangePointResult",
|
|
35
|
+
"ConfigurationError",
|
|
36
|
+
"DAGAnalysis",
|
|
37
|
+
"DAGAnalysisSummary",
|
|
38
|
+
"DriftResult",
|
|
39
|
+
"FlowSenseError",
|
|
40
|
+
"ImpactClassification",
|
|
41
|
+
"InsufficientHistoryError",
|
|
42
|
+
"InvalidTaskTimingError",
|
|
43
|
+
"MappedTaskAggregation",
|
|
44
|
+
"PropagationResult",
|
|
45
|
+
"RootCauseResult",
|
|
46
|
+
"Severity",
|
|
47
|
+
"TaskImpact",
|
|
48
|
+
"TaskRun",
|
|
49
|
+
"TrendDirection",
|
|
50
|
+
"TrendResult",
|
|
51
|
+
"severity_meets_threshold",
|
|
52
|
+
]
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
from enum import StrEnum
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class Severity(StrEnum):
|
|
5
|
+
NORMAL = "NORMAL"
|
|
6
|
+
MEDIUM = "MEDIUM"
|
|
7
|
+
HIGH = "HIGH"
|
|
8
|
+
CRITICAL = "CRITICAL"
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ImpactClassification(StrEnum):
|
|
12
|
+
NORMAL = "NORMAL"
|
|
13
|
+
OWN_DRIFT = "OWN_DRIFT"
|
|
14
|
+
INHERITED_DELAY = "INHERITED_DELAY"
|
|
15
|
+
COMBINED = "COMBINED"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class MappedTaskAggregation(StrEnum):
|
|
19
|
+
MAX = "MAX"
|
|
20
|
+
MEAN = "MEAN"
|
|
21
|
+
SUM = "SUM"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ChangeDirection(StrEnum):
|
|
25
|
+
INCREASE = "INCREASE"
|
|
26
|
+
DECREASE = "DECREASE"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class TrendDirection(StrEnum):
|
|
30
|
+
INCREASING = "INCREASING"
|
|
31
|
+
DECREASING = "DECREASING"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
SEVERITY_SCORE: dict[Severity, int] = {
|
|
35
|
+
Severity.NORMAL: 0,
|
|
36
|
+
Severity.MEDIUM: 1,
|
|
37
|
+
Severity.HIGH: 2,
|
|
38
|
+
Severity.CRITICAL: 3,
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def severity_meets_threshold(
|
|
43
|
+
severity: Severity,
|
|
44
|
+
threshold: Severity,
|
|
45
|
+
) -> bool:
|
|
46
|
+
"""Return whether a severity is at or above the requested threshold."""
|
|
47
|
+
return SEVERITY_SCORE[severity] >= SEVERITY_SCORE[threshold]
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
class FlowSenseError(Exception):
|
|
2
|
+
"""Base exception for expected FlowSense failures."""
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class ConfigurationError(FlowSenseError, ValueError):
|
|
6
|
+
"""Raised when FlowSense configuration is missing or invalid."""
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class InsufficientHistoryError(FlowSenseError, ValueError):
|
|
10
|
+
def __init__(
|
|
11
|
+
self,
|
|
12
|
+
subject_id: str,
|
|
13
|
+
required: int,
|
|
14
|
+
actual: int,
|
|
15
|
+
) -> None:
|
|
16
|
+
self.subject_id = subject_id
|
|
17
|
+
self.required = required
|
|
18
|
+
self.actual = actual
|
|
19
|
+
super().__init__(
|
|
20
|
+
f"{subject_id} için drift hesaplamak için en az {required} run gerekli."
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class InvalidTaskTimingError(FlowSenseError, ValueError):
|
|
25
|
+
"""Raised when task timestamps cannot produce a valid handoff timing."""
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
from datetime import datetime
|
|
2
|
+
|
|
3
|
+
from pydantic import BaseModel
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class TaskRun(BaseModel):
|
|
7
|
+
dag_id: str
|
|
8
|
+
dag_run_id: str
|
|
9
|
+
task_id: str
|
|
10
|
+
|
|
11
|
+
state: str | None = None
|
|
12
|
+
|
|
13
|
+
start_date: datetime | None = None
|
|
14
|
+
end_date: datetime | None = None
|
|
15
|
+
|
|
16
|
+
duration: float | None = None
|
|
17
|
+
try_number: int = 0
|
|
18
|
+
|
|
19
|
+
map_index: int = -1
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
|
|
3
|
+
from flowsense.domain.enums import MappedTaskAggregation
|
|
4
|
+
from flowsense.domain.exceptions import ConfigurationError
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclass(frozen=True)
|
|
8
|
+
class AnalysisPolicy:
|
|
9
|
+
minimum_history: int = 5
|
|
10
|
+
baseline_window: int | None = None
|
|
11
|
+
medium_threshold: float = 2.0
|
|
12
|
+
high_threshold: float = 3.5
|
|
13
|
+
critical_threshold: float = 5.0
|
|
14
|
+
mapped_task_aggregation: MappedTaskAggregation = MappedTaskAggregation.MAX
|
|
15
|
+
change_point_detection_enabled: bool = True
|
|
16
|
+
change_point_minimum_segment_size: int = 3
|
|
17
|
+
change_point_score_threshold: float = 3.5
|
|
18
|
+
trend_detection_enabled: bool = True
|
|
19
|
+
trend_minimum_observations: int = 5
|
|
20
|
+
trend_score_threshold: float = 3.5
|
|
21
|
+
trend_minimum_directional_consistency: float = 0.6
|
|
22
|
+
|
|
23
|
+
def __post_init__(self) -> None:
|
|
24
|
+
if self.minimum_history < 2:
|
|
25
|
+
raise ConfigurationError("minimum_history must be at least 2.")
|
|
26
|
+
if self.baseline_window is not None:
|
|
27
|
+
if self.baseline_window < 1:
|
|
28
|
+
raise ConfigurationError("baseline_window must be at least 1.")
|
|
29
|
+
if self.baseline_window < self.minimum_history - 1:
|
|
30
|
+
raise ConfigurationError(
|
|
31
|
+
"baseline_window must contain enough values for minimum_history."
|
|
32
|
+
)
|
|
33
|
+
if not (
|
|
34
|
+
0 < self.medium_threshold < self.high_threshold < self.critical_threshold
|
|
35
|
+
):
|
|
36
|
+
raise ConfigurationError(
|
|
37
|
+
"Severity thresholds must be positive and increasing."
|
|
38
|
+
)
|
|
39
|
+
if self.change_point_minimum_segment_size < 2:
|
|
40
|
+
raise ConfigurationError(
|
|
41
|
+
"change_point_minimum_segment_size must be at least 2."
|
|
42
|
+
)
|
|
43
|
+
if self.change_point_score_threshold <= 0:
|
|
44
|
+
raise ConfigurationError("change_point_score_threshold must be positive.")
|
|
45
|
+
if self.trend_minimum_observations < 3:
|
|
46
|
+
raise ConfigurationError("trend_minimum_observations must be at least 3.")
|
|
47
|
+
if self.trend_score_threshold <= 0:
|
|
48
|
+
raise ConfigurationError("trend_score_threshold must be positive.")
|
|
49
|
+
if not 0 < self.trend_minimum_directional_consistency <= 1:
|
|
50
|
+
raise ConfigurationError(
|
|
51
|
+
"trend_minimum_directional_consistency must be in (0, 1]."
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
DEFAULT_ANALYSIS_POLICY = AnalysisPolicy()
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
|
|
5
|
+
from flowsense.domain.enums import (
|
|
6
|
+
ChangeDirection,
|
|
7
|
+
ImpactClassification,
|
|
8
|
+
Severity,
|
|
9
|
+
TrendDirection,
|
|
10
|
+
)
|
|
11
|
+
from flowsense.domain.policy import DEFAULT_ANALYSIS_POLICY, AnalysisPolicy
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class DriftResult:
|
|
16
|
+
task_id: str
|
|
17
|
+
baseline: float
|
|
18
|
+
current: float
|
|
19
|
+
mad: float
|
|
20
|
+
robust_z_score: float
|
|
21
|
+
deviation_percent: float
|
|
22
|
+
severity: Severity
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True)
|
|
26
|
+
class ChangePointResult:
|
|
27
|
+
subject_id: str
|
|
28
|
+
change_index: int
|
|
29
|
+
before_median: float
|
|
30
|
+
after_median: float
|
|
31
|
+
change_percent: float | None
|
|
32
|
+
score: float
|
|
33
|
+
direction: ChangeDirection
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True)
|
|
37
|
+
class TrendResult:
|
|
38
|
+
subject_id: str
|
|
39
|
+
direction: TrendDirection
|
|
40
|
+
slope_per_observation: float
|
|
41
|
+
estimated_change: float
|
|
42
|
+
change_percent: float | None
|
|
43
|
+
score: float
|
|
44
|
+
directional_consistency: float
|
|
45
|
+
observations: int
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass(frozen=True)
|
|
49
|
+
class TaskImpact:
|
|
50
|
+
task_id: str
|
|
51
|
+
classification: ImpactClassification
|
|
52
|
+
task_severity: Severity
|
|
53
|
+
upstream_handoff_severity: Severity | None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass(frozen=True)
|
|
57
|
+
class PropagationResult:
|
|
58
|
+
origin_task: str
|
|
59
|
+
affected_tasks: list[str]
|
|
60
|
+
path: list[str]
|
|
61
|
+
propagation_score: float
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass(frozen=True)
|
|
65
|
+
class RootCauseResult:
|
|
66
|
+
task_id: str
|
|
67
|
+
classification: ImpactClassification
|
|
68
|
+
severity: Severity
|
|
69
|
+
propagation_score: float
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True)
|
|
73
|
+
class AnalysisDiagnostic:
|
|
74
|
+
code: str
|
|
75
|
+
subject_id: str
|
|
76
|
+
message: str
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@dataclass(frozen=True)
|
|
80
|
+
class DAGAnalysisSummary:
|
|
81
|
+
total_tasks: int
|
|
82
|
+
analyzed_tasks: int
|
|
83
|
+
analysis_coverage_percent: float
|
|
84
|
+
normal_tasks: int
|
|
85
|
+
medium_tasks: int
|
|
86
|
+
high_tasks: int
|
|
87
|
+
critical_tasks: int
|
|
88
|
+
anomalous_tasks: int
|
|
89
|
+
anomalous_handoffs: int
|
|
90
|
+
affected_tasks: int
|
|
91
|
+
change_points: int
|
|
92
|
+
trends: int
|
|
93
|
+
diagnostics: int
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@dataclass
|
|
97
|
+
class DAGAnalysis:
|
|
98
|
+
dag_id: str
|
|
99
|
+
runs_analyzed: int
|
|
100
|
+
overall_severity: Severity
|
|
101
|
+
primary_origin: RootCauseResult | None
|
|
102
|
+
drift_results: dict[str, DriftResult]
|
|
103
|
+
handoff_drift_results: dict[tuple[str, str], DriftResult]
|
|
104
|
+
task_impacts: dict[str, TaskImpact]
|
|
105
|
+
propagation_results: list[PropagationResult]
|
|
106
|
+
dependencies: dict[str, list[str]]
|
|
107
|
+
diagnostics: list[AnalysisDiagnostic] = field(default_factory=list)
|
|
108
|
+
policy: AnalysisPolicy = DEFAULT_ANALYSIS_POLICY
|
|
109
|
+
change_point_results: dict[str, ChangePointResult] = field(default_factory=dict)
|
|
110
|
+
handoff_change_point_results: dict[tuple[str, str], ChangePointResult] = field(
|
|
111
|
+
default_factory=dict
|
|
112
|
+
)
|
|
113
|
+
trend_results: dict[str, TrendResult] = field(default_factory=dict)
|
|
114
|
+
handoff_trend_results: dict[tuple[str, str], TrendResult] = field(
|
|
115
|
+
default_factory=dict
|
|
116
|
+
)
|
|
117
|
+
current_dag_run_id: str | None = None
|
|
118
|
+
|
|
119
|
+
@property
|
|
120
|
+
def summary(self) -> DAGAnalysisSummary:
|
|
121
|
+
task_ids = {
|
|
122
|
+
*self.dependencies,
|
|
123
|
+
*(task for tasks in self.dependencies.values() for task in tasks),
|
|
124
|
+
*self.drift_results,
|
|
125
|
+
*self.task_impacts,
|
|
126
|
+
*self.change_point_results,
|
|
127
|
+
*self.trend_results,
|
|
128
|
+
*(
|
|
129
|
+
task
|
|
130
|
+
for edge in (
|
|
131
|
+
*self.handoff_drift_results,
|
|
132
|
+
*self.handoff_change_point_results,
|
|
133
|
+
*self.handoff_trend_results,
|
|
134
|
+
)
|
|
135
|
+
for task in edge
|
|
136
|
+
),
|
|
137
|
+
*(
|
|
138
|
+
task
|
|
139
|
+
for result in self.propagation_results
|
|
140
|
+
for task in (result.origin_task, *result.affected_tasks)
|
|
141
|
+
),
|
|
142
|
+
}
|
|
143
|
+
severity_counts = {
|
|
144
|
+
severity: sum(
|
|
145
|
+
result.severity == severity for result in self.drift_results.values()
|
|
146
|
+
)
|
|
147
|
+
for severity in Severity
|
|
148
|
+
}
|
|
149
|
+
analyzed_tasks = len(self.drift_results)
|
|
150
|
+
total_tasks = len(task_ids)
|
|
151
|
+
anomalous_severities = {
|
|
152
|
+
Severity.MEDIUM,
|
|
153
|
+
Severity.HIGH,
|
|
154
|
+
Severity.CRITICAL,
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
return DAGAnalysisSummary(
|
|
158
|
+
total_tasks=total_tasks,
|
|
159
|
+
analyzed_tasks=analyzed_tasks,
|
|
160
|
+
analysis_coverage_percent=(
|
|
161
|
+
analyzed_tasks / total_tasks * 100 if total_tasks else 0.0
|
|
162
|
+
),
|
|
163
|
+
normal_tasks=severity_counts[Severity.NORMAL],
|
|
164
|
+
medium_tasks=severity_counts[Severity.MEDIUM],
|
|
165
|
+
high_tasks=severity_counts[Severity.HIGH],
|
|
166
|
+
critical_tasks=severity_counts[Severity.CRITICAL],
|
|
167
|
+
anomalous_tasks=sum(
|
|
168
|
+
result.severity in anomalous_severities
|
|
169
|
+
for result in self.drift_results.values()
|
|
170
|
+
),
|
|
171
|
+
anomalous_handoffs=sum(
|
|
172
|
+
result.severity in anomalous_severities
|
|
173
|
+
for result in self.handoff_drift_results.values()
|
|
174
|
+
),
|
|
175
|
+
affected_tasks=len(
|
|
176
|
+
{
|
|
177
|
+
task
|
|
178
|
+
for result in self.propagation_results
|
|
179
|
+
for task in result.affected_tasks
|
|
180
|
+
}
|
|
181
|
+
),
|
|
182
|
+
change_points=(
|
|
183
|
+
len(self.change_point_results) + len(self.handoff_change_point_results)
|
|
184
|
+
),
|
|
185
|
+
trends=len(self.trend_results) + len(self.handoff_trend_results),
|
|
186
|
+
diagnostics=len(self.diagnostics),
|
|
187
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Backward-compatible analyzer entry point.
|
|
2
|
+
|
|
3
|
+
New integrations should inject a data source into
|
|
4
|
+
``flowsense.application.analyze_dag``.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from flowsense.application import analyze_dag as analyze_dag_with_source
|
|
8
|
+
from flowsense.domain import DAGAnalysis
|
|
9
|
+
from flowsense.infrastructure.airflow import AirflowClient
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def analyze_dag(dag_id: str) -> DAGAnalysis:
|
|
13
|
+
with AirflowClient() as source:
|
|
14
|
+
return analyze_dag_with_source(
|
|
15
|
+
dag_id=dag_id,
|
|
16
|
+
source=source,
|
|
17
|
+
)
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
|
|
7
|
+
from flowsense.domain import ChangeDirection, ChangePointResult
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _candidate_score(
|
|
11
|
+
before: np.ndarray, after: np.ndarray
|
|
12
|
+
) -> tuple[float, float, float]:
|
|
13
|
+
before_median = float(np.median(before))
|
|
14
|
+
after_median = float(np.median(after))
|
|
15
|
+
difference = after_median - before_median
|
|
16
|
+
|
|
17
|
+
residuals = np.concatenate(
|
|
18
|
+
(
|
|
19
|
+
np.abs(before - before_median),
|
|
20
|
+
np.abs(after - after_median),
|
|
21
|
+
)
|
|
22
|
+
)
|
|
23
|
+
noise = float(np.median(residuals))
|
|
24
|
+
|
|
25
|
+
if math.isclose(noise, 0.0):
|
|
26
|
+
score = 0.0 if math.isclose(difference, 0.0) else 5.0
|
|
27
|
+
else:
|
|
28
|
+
score = abs(0.6745 * difference / noise)
|
|
29
|
+
|
|
30
|
+
return score, before_median, after_median
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def detect_change_point(
|
|
34
|
+
subject_id: str,
|
|
35
|
+
values: list[float],
|
|
36
|
+
*,
|
|
37
|
+
minimum_segment_size: int = 3,
|
|
38
|
+
score_threshold: float = 3.5,
|
|
39
|
+
) -> ChangePointResult | None:
|
|
40
|
+
"""Detect the strongest persistent level shift in an ordered time series."""
|
|
41
|
+
if minimum_segment_size < 2:
|
|
42
|
+
raise ValueError("minimum_segment_size must be at least 2")
|
|
43
|
+
|
|
44
|
+
if score_threshold <= 0:
|
|
45
|
+
raise ValueError("score_threshold must be positive")
|
|
46
|
+
|
|
47
|
+
if len(values) < minimum_segment_size * 2:
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
observations = np.asarray(values, dtype=float)
|
|
51
|
+
best_candidate: tuple[float, int, float, float] | None = None
|
|
52
|
+
|
|
53
|
+
for change_index in range(
|
|
54
|
+
minimum_segment_size,
|
|
55
|
+
len(observations) - minimum_segment_size + 1,
|
|
56
|
+
):
|
|
57
|
+
score, before_median, after_median = _candidate_score(
|
|
58
|
+
observations[:change_index],
|
|
59
|
+
observations[change_index:],
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
if best_candidate is None or score > best_candidate[0]:
|
|
63
|
+
best_candidate = (score, change_index, before_median, after_median)
|
|
64
|
+
|
|
65
|
+
if best_candidate is None or best_candidate[0] < score_threshold:
|
|
66
|
+
return None
|
|
67
|
+
|
|
68
|
+
score, change_index, before_median, after_median = best_candidate
|
|
69
|
+
change_percent = (
|
|
70
|
+
(after_median - before_median) / before_median * 100
|
|
71
|
+
if not math.isclose(before_median, 0.0)
|
|
72
|
+
else None
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
return ChangePointResult(
|
|
76
|
+
subject_id=subject_id,
|
|
77
|
+
change_index=change_index,
|
|
78
|
+
before_median=before_median,
|
|
79
|
+
after_median=after_median,
|
|
80
|
+
change_percent=change_percent,
|
|
81
|
+
score=score,
|
|
82
|
+
direction=(
|
|
83
|
+
ChangeDirection.INCREASE
|
|
84
|
+
if after_median > before_median
|
|
85
|
+
else ChangeDirection.DECREASE
|
|
86
|
+
),
|
|
87
|
+
)
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
|
|
7
|
+
from flowsense.domain import (
|
|
8
|
+
DEFAULT_ANALYSIS_POLICY,
|
|
9
|
+
AnalysisPolicy,
|
|
10
|
+
DriftResult,
|
|
11
|
+
InsufficientHistoryError,
|
|
12
|
+
Severity,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def calculate_drift(
|
|
17
|
+
task_id: str,
|
|
18
|
+
durations: list[float],
|
|
19
|
+
policy: AnalysisPolicy = DEFAULT_ANALYSIS_POLICY,
|
|
20
|
+
) -> DriftResult:
|
|
21
|
+
if len(durations) < policy.minimum_history:
|
|
22
|
+
raise InsufficientHistoryError(
|
|
23
|
+
subject_id=task_id,
|
|
24
|
+
required=policy.minimum_history,
|
|
25
|
+
actual=len(durations),
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
baseline_durations = durations[:-1]
|
|
29
|
+
if policy.baseline_window is not None:
|
|
30
|
+
baseline_durations = baseline_durations[-policy.baseline_window :]
|
|
31
|
+
|
|
32
|
+
baseline_values = np.array(
|
|
33
|
+
baseline_durations,
|
|
34
|
+
dtype=float,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
current = float(durations[-1])
|
|
38
|
+
|
|
39
|
+
median = float(np.median(baseline_values))
|
|
40
|
+
|
|
41
|
+
absolute_deviations = np.abs(baseline_values - median)
|
|
42
|
+
|
|
43
|
+
mad = float(np.median(absolute_deviations))
|
|
44
|
+
|
|
45
|
+
if mad == 0:
|
|
46
|
+
if math.isclose(current, median):
|
|
47
|
+
robust_z_score = 0.0
|
|
48
|
+
else:
|
|
49
|
+
robust_z_score = math.copysign(
|
|
50
|
+
policy.critical_threshold,
|
|
51
|
+
current - median,
|
|
52
|
+
)
|
|
53
|
+
else:
|
|
54
|
+
robust_z_score = 0.6745 * (current - median) / mad
|
|
55
|
+
|
|
56
|
+
if median == 0:
|
|
57
|
+
deviation_percent = 0.0
|
|
58
|
+
else:
|
|
59
|
+
deviation_percent = (current - median) / median * 100
|
|
60
|
+
|
|
61
|
+
absolute_z = abs(robust_z_score)
|
|
62
|
+
|
|
63
|
+
if absolute_z >= policy.critical_threshold:
|
|
64
|
+
severity = Severity.CRITICAL
|
|
65
|
+
elif absolute_z >= policy.high_threshold:
|
|
66
|
+
severity = Severity.HIGH
|
|
67
|
+
elif absolute_z >= policy.medium_threshold:
|
|
68
|
+
severity = Severity.MEDIUM
|
|
69
|
+
else:
|
|
70
|
+
severity = Severity.NORMAL
|
|
71
|
+
|
|
72
|
+
return DriftResult(
|
|
73
|
+
task_id=task_id,
|
|
74
|
+
baseline=median,
|
|
75
|
+
current=current,
|
|
76
|
+
mad=mad,
|
|
77
|
+
robust_z_score=robust_z_score,
|
|
78
|
+
deviation_percent=deviation_percent,
|
|
79
|
+
severity=severity,
|
|
80
|
+
)
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections import defaultdict
|
|
4
|
+
|
|
5
|
+
from flowsense.domain import MappedTaskAggregation, TaskRun
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def build_duration_history(
|
|
9
|
+
task_runs: list[TaskRun],
|
|
10
|
+
aggregation: MappedTaskAggregation = MappedTaskAggregation.MAX,
|
|
11
|
+
) -> dict[str, list[float]]:
|
|
12
|
+
"""Build logical-task history using the slowest mapped instance per DAG run."""
|
|
13
|
+
grouped_durations: dict[tuple[str, str], list[float]] = defaultdict(list)
|
|
14
|
+
|
|
15
|
+
for task_run in task_runs:
|
|
16
|
+
if task_run.state != "success" or task_run.duration is None:
|
|
17
|
+
continue
|
|
18
|
+
|
|
19
|
+
key = (task_run.dag_run_id, task_run.task_id)
|
|
20
|
+
grouped_durations[key].append(task_run.duration)
|
|
21
|
+
|
|
22
|
+
def aggregate(values: list[float]) -> float:
|
|
23
|
+
if aggregation is MappedTaskAggregation.SUM:
|
|
24
|
+
return sum(values)
|
|
25
|
+
if aggregation is MappedTaskAggregation.MEAN:
|
|
26
|
+
return sum(values) / len(values)
|
|
27
|
+
return max(values)
|
|
28
|
+
|
|
29
|
+
history: dict[str, list[float]] = defaultdict(list)
|
|
30
|
+
|
|
31
|
+
for (_dag_run_id, task_id), durations in grouped_durations.items():
|
|
32
|
+
history[task_id].append(aggregate(durations))
|
|
33
|
+
|
|
34
|
+
return dict(history)
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from flowsense.domain import (
|
|
4
|
+
DriftResult,
|
|
5
|
+
ImpactClassification,
|
|
6
|
+
Severity,
|
|
7
|
+
TaskImpact,
|
|
8
|
+
)
|
|
9
|
+
from flowsense.domain.enums import SEVERITY_SCORE
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def classify_task_impact(
|
|
13
|
+
task_id: str,
|
|
14
|
+
task_drift: DriftResult,
|
|
15
|
+
upstream_handoff_drifts: list[DriftResult],
|
|
16
|
+
) -> TaskImpact:
|
|
17
|
+
anomalous_handoffs = [
|
|
18
|
+
drift
|
|
19
|
+
for drift in upstream_handoff_drifts
|
|
20
|
+
if drift.severity in {Severity.MEDIUM, Severity.HIGH, Severity.CRITICAL}
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
task_is_anomalous = task_drift.severity in {
|
|
24
|
+
Severity.MEDIUM,
|
|
25
|
+
Severity.HIGH,
|
|
26
|
+
Severity.CRITICAL,
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
if task_is_anomalous and anomalous_handoffs:
|
|
30
|
+
classification = ImpactClassification.COMBINED
|
|
31
|
+
elif task_is_anomalous:
|
|
32
|
+
classification = ImpactClassification.OWN_DRIFT
|
|
33
|
+
elif anomalous_handoffs:
|
|
34
|
+
classification = ImpactClassification.INHERITED_DELAY
|
|
35
|
+
else:
|
|
36
|
+
classification = ImpactClassification.NORMAL
|
|
37
|
+
|
|
38
|
+
upstream_handoff_severity = None
|
|
39
|
+
|
|
40
|
+
if anomalous_handoffs:
|
|
41
|
+
upstream_handoff_severity = max(
|
|
42
|
+
anomalous_handoffs,
|
|
43
|
+
key=lambda result: SEVERITY_SCORE[result.severity],
|
|
44
|
+
).severity
|
|
45
|
+
|
|
46
|
+
return TaskImpact(
|
|
47
|
+
task_id=task_id,
|
|
48
|
+
classification=classification,
|
|
49
|
+
task_severity=task_drift.severity,
|
|
50
|
+
upstream_handoff_severity=upstream_handoff_severity,
|
|
51
|
+
)
|