tketool.pipeline 1.3.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tketool_pipeline-1.3.5/PKG-INFO +49 -0
- tketool_pipeline-1.3.5/README.md +31 -0
- tketool_pipeline-1.3.5/pyproject.toml +31 -0
- tketool_pipeline-1.3.5/setup.cfg +4 -0
- tketool_pipeline-1.3.5/src/tketool/pipeline/__init__.py +19 -0
- tketool_pipeline-1.3.5/src/tketool/pipeline/execution.py +218 -0
- tketool_pipeline-1.3.5/src/tketool/pipeline/pipeline.py +386 -0
- tketool_pipeline-1.3.5/src/tketool.pipeline.egg-info/PKG-INFO +49 -0
- tketool_pipeline-1.3.5/src/tketool.pipeline.egg-info/SOURCES.txt +10 -0
- tketool_pipeline-1.3.5/src/tketool.pipeline.egg-info/dependency_links.txt +1 -0
- tketool_pipeline-1.3.5/src/tketool.pipeline.egg-info/requires.txt +4 -0
- tketool_pipeline-1.3.5/src/tketool.pipeline.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tketool.pipeline
|
|
3
|
+
Version: 1.3.5
|
|
4
|
+
Summary: Stage-based processing pipeline and injection decorators for tketool
|
|
5
|
+
Author-email: Ke <jiangke1207@icloud.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://pypi.org/project/tketool.pipeline/
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Python: >=3.10
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: tketool.core[console]==1.3.5
|
|
16
|
+
Provides-Extra: test
|
|
17
|
+
Requires-Dist: pytest<9,>=8; extra == "test"
|
|
18
|
+
|
|
19
|
+
# tketool.pipeline
|
|
20
|
+
|
|
21
|
+
Small stage-based processing pipelines with ordered decorators, optional
|
|
22
|
+
parallel item processing, and local buffering.
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install tketool.pipeline
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
```python
|
|
29
|
+
from enum import Enum
|
|
30
|
+
|
|
31
|
+
from tketool.pipeline import Pipeline, invoke_all
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class Stage(Enum):
|
|
35
|
+
PREPARE = "prepare"
|
|
36
|
+
RUN = "run"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Steps:
|
|
40
|
+
@invoke_all(Stage.PREPARE.value, priority=10)
|
|
41
|
+
def prepare(self, board):
|
|
42
|
+
board.current_data = {"ready": True}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
pipeline = Pipeline(Stage, acts=[Steps()])
|
|
46
|
+
result = pipeline()
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
The retired `tketool.injections` path is not shipped.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# tketool.pipeline
|
|
2
|
+
|
|
3
|
+
Small stage-based processing pipelines with ordered decorators, optional
|
|
4
|
+
parallel item processing, and local buffering.
|
|
5
|
+
|
|
6
|
+
```bash
|
|
7
|
+
pip install tketool.pipeline
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
from enum import Enum
|
|
12
|
+
|
|
13
|
+
from tketool.pipeline import Pipeline, invoke_all
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class Stage(Enum):
|
|
17
|
+
PREPARE = "prepare"
|
|
18
|
+
RUN = "run"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class Steps:
|
|
22
|
+
@invoke_all(Stage.PREPARE.value, priority=10)
|
|
23
|
+
def prepare(self, board):
|
|
24
|
+
board.current_data = {"ready": True}
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
pipeline = Pipeline(Stage, acts=[Steps()])
|
|
28
|
+
result = pipeline()
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
The retired `tketool.injections` path is not shipped.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "tketool.pipeline"
|
|
7
|
+
version = "1.3.5"
|
|
8
|
+
description = "Stage-based processing pipeline and injection decorators for tketool"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [{ name = "Ke", email = "jiangke1207@icloud.com" }]
|
|
13
|
+
dependencies = ["tketool.core[console]==1.3.5"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"Programming Language :: Python :: 3.10",
|
|
17
|
+
"Programming Language :: Python :: 3.11",
|
|
18
|
+
"Programming Language :: Python :: 3.12",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
[project.optional-dependencies]
|
|
23
|
+
test = ["pytest>=8,<9"]
|
|
24
|
+
|
|
25
|
+
[project.urls]
|
|
26
|
+
Homepage = "https://pypi.org/project/tketool.pipeline/"
|
|
27
|
+
|
|
28
|
+
[tool.setuptools.packages.find]
|
|
29
|
+
where = ["src"]
|
|
30
|
+
include = ["tketool*"]
|
|
31
|
+
namespaces = true
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Stage-based processing pipeline and execution decorators."""
|
|
2
|
+
|
|
3
|
+
from .execution import (
|
|
4
|
+
CrossStageCopy,
|
|
5
|
+
ExecutionBoard,
|
|
6
|
+
invoke_all,
|
|
7
|
+
invoke_one,
|
|
8
|
+
invoke_per_stage,
|
|
9
|
+
)
|
|
10
|
+
from .pipeline import Pipeline
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"CrossStageCopy",
|
|
14
|
+
"ExecutionBoard",
|
|
15
|
+
"Pipeline",
|
|
16
|
+
"invoke_all",
|
|
17
|
+
"invoke_one",
|
|
18
|
+
"invoke_per_stage",
|
|
19
|
+
]
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
import functools, abc
|
|
2
|
+
from enum import Enum
|
|
3
|
+
from tketool.core.progress import ProgressStatusBar
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def invoke_all(stage, priority: int, buffer=False, enable=True):
|
|
7
|
+
"""
|
|
8
|
+
装饰器函数,用于标记需要在特定阶段执行的所有方法
|
|
9
|
+
|
|
10
|
+
Args:
|
|
11
|
+
stage: 执行阶段
|
|
12
|
+
priority (int): 执行优先级
|
|
13
|
+
buffer (bool): 是否启用缓冲
|
|
14
|
+
enable (bool): 是否启用该方法
|
|
15
|
+
|
|
16
|
+
Returns:
|
|
17
|
+
wrapper: 装饰后的函数
|
|
18
|
+
"""
|
|
19
|
+
def decorator(func):
|
|
20
|
+
@functools.wraps(func)
|
|
21
|
+
def wrapper(*args, **kwargs):
|
|
22
|
+
try:
|
|
23
|
+
return func(*args, **kwargs)
|
|
24
|
+
except Exception as ex:
|
|
25
|
+
return ex
|
|
26
|
+
|
|
27
|
+
wrapper._priority = priority
|
|
28
|
+
wrapper._stage = stage
|
|
29
|
+
wrapper._buffer = buffer
|
|
30
|
+
wrapper._enable = enable
|
|
31
|
+
return wrapper
|
|
32
|
+
|
|
33
|
+
return decorator
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def invoke_per_stage(priority: int, buffer=False, enable=True):
|
|
37
|
+
"""
|
|
38
|
+
装饰器函数,用于标记每个阶段都需要执行的方法
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
priority (int): 执行优先级
|
|
42
|
+
buffer (bool): 是否启用缓冲
|
|
43
|
+
enable (bool): 是否启用该方法
|
|
44
|
+
|
|
45
|
+
Returns:
|
|
46
|
+
wrapper: 装饰后的函数
|
|
47
|
+
"""
|
|
48
|
+
def decorator(func):
|
|
49
|
+
@functools.wraps(func)
|
|
50
|
+
def wrapper(*args, **kwargs):
|
|
51
|
+
return func(*args, **kwargs)
|
|
52
|
+
|
|
53
|
+
wrapper._priority = priority
|
|
54
|
+
wrapper._buffer = buffer
|
|
55
|
+
wrapper._enable = enable
|
|
56
|
+
return wrapper
|
|
57
|
+
|
|
58
|
+
return decorator
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def invoke_one(stage, priority: int, enum_key: str = None, parallel: bool = False,
|
|
62
|
+
buffer=False, enable=True, pass_to=0):
|
|
63
|
+
"""
|
|
64
|
+
装饰器函数,用于标记在特定阶段只执行一次的方法
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
stage: 执行阶段
|
|
68
|
+
priority (int): 执行优先级
|
|
69
|
+
enum_key (str): 枚举键名
|
|
70
|
+
parallel (bool): 是否允许并行执行
|
|
71
|
+
buffer (bool): 是否启用缓冲
|
|
72
|
+
enable (bool): 是否启用该方法
|
|
73
|
+
pass_to (int): 传递目标
|
|
74
|
+
|
|
75
|
+
Returns:
|
|
76
|
+
wrapper: 装饰后的函数
|
|
77
|
+
"""
|
|
78
|
+
def decorator(func):
|
|
79
|
+
@functools.wraps(func)
|
|
80
|
+
def wrapper(*args, **kwargs):
|
|
81
|
+
return func(*args, **kwargs)
|
|
82
|
+
|
|
83
|
+
wrapper._priority = priority
|
|
84
|
+
wrapper._enum_key = enum_key
|
|
85
|
+
wrapper._stage = stage
|
|
86
|
+
wrapper._parallel = parallel
|
|
87
|
+
wrapper._buffer = buffer
|
|
88
|
+
wrapper._enable = enable
|
|
89
|
+
wrapper._pass_to = pass_to
|
|
90
|
+
return wrapper
|
|
91
|
+
|
|
92
|
+
return decorator
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class CrossStageCopy:
|
|
96
|
+
|
|
97
|
+
@invoke_per_stage(priority=0)
|
|
98
|
+
def cross_copy(self, board):
|
|
99
|
+
"""
|
|
100
|
+
在不同阶段间复制数据的方法
|
|
101
|
+
|
|
102
|
+
Args:
|
|
103
|
+
board: 数据面板对象,包含当前阶段和上一阶段的数据
|
|
104
|
+
"""
|
|
105
|
+
if board.last_data is None:
|
|
106
|
+
return
|
|
107
|
+
elif isinstance(board.last_data, list):
|
|
108
|
+
board.current_data = []
|
|
109
|
+
for item in board.last_data:
|
|
110
|
+
board.current_data.append(item)
|
|
111
|
+
elif isinstance(board.last_data, dict):
|
|
112
|
+
board.current_data = {}
|
|
113
|
+
for k, v in board.last_data.items():
|
|
114
|
+
board.current_data[k] = v
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class ExecutionBoard:
|
|
118
|
+
"""
|
|
119
|
+
用于管理执行阶段和数据的面板类
|
|
120
|
+
"""
|
|
121
|
+
|
|
122
|
+
def __init__(self):
|
|
123
|
+
"""
|
|
124
|
+
初始化执行面板
|
|
125
|
+
"""
|
|
126
|
+
self.current_stage = None
|
|
127
|
+
self.last_stage = None
|
|
128
|
+
self.process_bar = ProgressStatusBar(progress_bar_length=10, max_str_length=200)
|
|
129
|
+
|
|
130
|
+
self.sub_process_info = []
|
|
131
|
+
self.oper_dict = {}
|
|
132
|
+
self.stages = {}
|
|
133
|
+
self.logs = []
|
|
134
|
+
self.additional_dict = {}
|
|
135
|
+
|
|
136
|
+
@property
|
|
137
|
+
def serialize_tuple(self):
|
|
138
|
+
"""
|
|
139
|
+
获取序列化数据元组
|
|
140
|
+
|
|
141
|
+
Returns:
|
|
142
|
+
tuple: 包含面板所有数据的元组
|
|
143
|
+
"""
|
|
144
|
+
return (self.sub_process_info, self.oper_dict, self.stages, self.logs, self.additional_dict)
|
|
145
|
+
|
|
146
|
+
@serialize_tuple.setter
|
|
147
|
+
def serialize_tuple(self, value):
|
|
148
|
+
"""
|
|
149
|
+
从序列化数据元组恢复面板数据
|
|
150
|
+
|
|
151
|
+
Args:
|
|
152
|
+
value: 包含面板数据的元组
|
|
153
|
+
"""
|
|
154
|
+
self.sub_process_info = value[0]
|
|
155
|
+
self.oper_dict = value[1]
|
|
156
|
+
self.stages = value[2]
|
|
157
|
+
self.logs = value[3]
|
|
158
|
+
self.additional_dict = value[4]
|
|
159
|
+
|
|
160
|
+
for k, v in self.additional_dict.items():
|
|
161
|
+
self.__setattr__(k, v)
|
|
162
|
+
|
|
163
|
+
def set_field(self, key, value):
|
|
164
|
+
"""
|
|
165
|
+
设置面板的附加字段
|
|
166
|
+
|
|
167
|
+
Args:
|
|
168
|
+
key: 字段名
|
|
169
|
+
value: 字段值
|
|
170
|
+
"""
|
|
171
|
+
self.additional_dict[key] = value
|
|
172
|
+
self.__setattr__(key, value)
|
|
173
|
+
|
|
174
|
+
def set_stage(self, stage):
|
|
175
|
+
"""
|
|
176
|
+
设置当前执行阶段
|
|
177
|
+
|
|
178
|
+
Args:
|
|
179
|
+
stage: 要设置的阶段
|
|
180
|
+
"""
|
|
181
|
+
if stage not in self.stages:
|
|
182
|
+
self.stages[stage] = {}
|
|
183
|
+
self.last_stage = self.current_stage
|
|
184
|
+
self.current_stage = stage
|
|
185
|
+
|
|
186
|
+
@property
|
|
187
|
+
def current_data(self):
|
|
188
|
+
"""
|
|
189
|
+
获取当前阶段的数据
|
|
190
|
+
|
|
191
|
+
Returns:
|
|
192
|
+
当前阶段的数据,如果当前阶段未设置则返回None
|
|
193
|
+
"""
|
|
194
|
+
if self.current_stage is None:
|
|
195
|
+
return None
|
|
196
|
+
return self.stages[self.current_stage]
|
|
197
|
+
|
|
198
|
+
@property
|
|
199
|
+
def last_data(self):
|
|
200
|
+
"""
|
|
201
|
+
获取上一阶段的数据
|
|
202
|
+
|
|
203
|
+
Returns:
|
|
204
|
+
上一阶段的数据,如果上一阶段未设置则返回None
|
|
205
|
+
"""
|
|
206
|
+
if self.last_stage is None:
|
|
207
|
+
return None
|
|
208
|
+
return self.stages[self.last_stage]
|
|
209
|
+
|
|
210
|
+
@current_data.setter
|
|
211
|
+
def current_data(self, v):
|
|
212
|
+
"""
|
|
213
|
+
设置当前阶段的数据
|
|
214
|
+
|
|
215
|
+
Args:
|
|
216
|
+
v: 要设置的数据
|
|
217
|
+
"""
|
|
218
|
+
self.stages[self.current_stage] = v
|
|
@@ -0,0 +1,386 @@
|
|
|
1
|
+
from tketool.pipeline.execution import ExecutionBoard
|
|
2
|
+
import time
|
|
3
|
+
from tketool.core.concurrency import execute_multitask
|
|
4
|
+
from tketool.core.cache.base import buffer_item, flush, get_buffer_item, has_item_key
|
|
5
|
+
from tketool.core.cache import shelve_many as _shelve_many
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Pipeline:
|
|
9
|
+
"""
|
|
10
|
+
多阶段处理管道类,支持按阶段执行不同的处理函数,具备并行处理和缓冲机制
|
|
11
|
+
|
|
12
|
+
主要功能:
|
|
13
|
+
- 支持多阶段串行执行
|
|
14
|
+
- 支持多线程并行处理
|
|
15
|
+
- 支持结果缓冲和恢复
|
|
16
|
+
- 支持按优先级排序执行
|
|
17
|
+
- 支持忽略指定函数
|
|
18
|
+
|
|
19
|
+
使用示例:
|
|
20
|
+
```python
|
|
21
|
+
from enum import Enum
|
|
22
|
+
|
|
23
|
+
class ProcessStages(Enum):
|
|
24
|
+
INIT = "init"
|
|
25
|
+
PROCESS = "process"
|
|
26
|
+
FINISH = "finish"
|
|
27
|
+
|
|
28
|
+
class MyProcessor:
|
|
29
|
+
@invoke_all("init", priority=1)
|
|
30
|
+
def setup(self, board):
|
|
31
|
+
board.current_data = {"status": "initialized"}
|
|
32
|
+
|
|
33
|
+
@invoke_one("process", priority=1, enum_key="items", parallel=True)
|
|
34
|
+
def process_item(self, item, board):
|
|
35
|
+
return f"processed_{item}"
|
|
36
|
+
|
|
37
|
+
processor = MyProcessor()
|
|
38
|
+
pipeline = Pipeline(ProcessStages, acts=[processor], thread_count=4)
|
|
39
|
+
result = pipeline(items=["item1", "item2", "item3"])
|
|
40
|
+
```
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
def __init__(self, stages, acts=[], ignore_funs=[], thread_count=1):
|
|
44
|
+
"""
|
|
45
|
+
初始化处理管道
|
|
46
|
+
|
|
47
|
+
Args:
|
|
48
|
+
stages: 处理阶段列表或枚举类
|
|
49
|
+
- 如果是枚举类,会自动提取所有成员名称
|
|
50
|
+
- 如果是列表,直接使用
|
|
51
|
+
acts (list): 处理类实例列表,包含用装饰器标记的处理方法
|
|
52
|
+
ignore_funs (list): 要忽略的函数名称列表,格式为"ClassName.method_name"
|
|
53
|
+
thread_count (int): 并行处理的线程数,默认为1
|
|
54
|
+
|
|
55
|
+
示例:
|
|
56
|
+
```python
|
|
57
|
+
# 使用枚举定义阶段
|
|
58
|
+
pipeline = Pipeline(MyStages, acts=[processor1, processor2], thread_count=4)
|
|
59
|
+
|
|
60
|
+
# 使用列表定义阶段
|
|
61
|
+
pipeline = Pipeline(["init", "process", "finish"], acts=[processor])
|
|
62
|
+
|
|
63
|
+
# 忽略特定方法
|
|
64
|
+
pipeline = Pipeline(stages, acts=[processor], ignore_funs=["MyProcessor.debug_method"])
|
|
65
|
+
```
|
|
66
|
+
"""
|
|
67
|
+
if isinstance(stages, list):
|
|
68
|
+
self.all_stages = stages
|
|
69
|
+
else:
|
|
70
|
+
# 从枚举类中提取成员值(而不是名称)
|
|
71
|
+
self.all_stages = [memb.value for name, memb in stages.__members__.items()]
|
|
72
|
+
|
|
73
|
+
self.multi_thread_count = thread_count
|
|
74
|
+
self.all_injection_class = acts
|
|
75
|
+
self.sub_processes = {k: [] for k in self.all_stages}
|
|
76
|
+
self.ignore_names = set(ignore_funs)
|
|
77
|
+
|
|
78
|
+
# 扫描所有处理类,提取标记的方法
|
|
79
|
+
for pl in self.all_injection_class:
|
|
80
|
+
for name, funs in self._get_process_map(pl).items():
|
|
81
|
+
if name in self.ignore_names:
|
|
82
|
+
continue
|
|
83
|
+
|
|
84
|
+
# 如果方法指定了特定阶段,只在该阶段执行
|
|
85
|
+
if 'stage' in funs:
|
|
86
|
+
if funs['stage'] in self.sub_processes:
|
|
87
|
+
self.sub_processes[funs['stage']].append(funs)
|
|
88
|
+
else:
|
|
89
|
+
# 否则在所有阶段都执行
|
|
90
|
+
for st in self.all_stages:
|
|
91
|
+
self.sub_processes[st].append(funs)
|
|
92
|
+
|
|
93
|
+
# 按优先级排序各阶段的处理函数
|
|
94
|
+
for key in self.sub_processes.keys():
|
|
95
|
+
self.sub_processes[key].sort(key=lambda x: x['priority'])
|
|
96
|
+
|
|
97
|
+
self.buffer_switch = True
|
|
98
|
+
|
|
99
|
+
def disable_buffer(self):
|
|
100
|
+
"""
|
|
101
|
+
禁用缓冲机制
|
|
102
|
+
|
|
103
|
+
调用此方法后,所有处理过程将不再使用缓冲,每次都会重新执行
|
|
104
|
+
|
|
105
|
+
示例:
|
|
106
|
+
```python
|
|
107
|
+
pipeline = Pipeline(stages, acts=[processor])
|
|
108
|
+
pipeline.disable_buffer() # 禁用缓冲
|
|
109
|
+
result = pipeline() # 不使用缓冲执行
|
|
110
|
+
```
|
|
111
|
+
"""
|
|
112
|
+
self.buffer_switch = False
|
|
113
|
+
|
|
114
|
+
def _get_process_map(self, target) -> dict:
|
|
115
|
+
"""
|
|
116
|
+
从目标对象中提取所有标记的处理方法
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
target: 要扫描的对象实例
|
|
120
|
+
|
|
121
|
+
Returns:
|
|
122
|
+
dict: 包含方法信息的字典,键为方法名,值为方法元数据
|
|
123
|
+
|
|
124
|
+
方法元数据包含:
|
|
125
|
+
- name: 方法名称
|
|
126
|
+
- type: 处理类型 ("one" 或 "all")
|
|
127
|
+
- stage: 执行阶段(可选)
|
|
128
|
+
- priority: 优先级
|
|
129
|
+
- func: 方法对象
|
|
130
|
+
- 其他装饰器参数
|
|
131
|
+
"""
|
|
132
|
+
run_list = {}
|
|
133
|
+
for name in dir(target):
|
|
134
|
+
method = getattr(target, name)
|
|
135
|
+
process_name = f"{type(target).__name__}.{name}"
|
|
136
|
+
|
|
137
|
+
# 检查方法是否有优先级标记(表示被装饰器处理过)
|
|
138
|
+
if hasattr(method, '_priority'):
|
|
139
|
+
method_info = {
|
|
140
|
+
'name': process_name,
|
|
141
|
+
'priority': method._priority,
|
|
142
|
+
'buffer': method._buffer,
|
|
143
|
+
'func': method,
|
|
144
|
+
'enable': method._enable,
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
# 检查是否是单项处理方法(invoke_one装饰的)
|
|
148
|
+
if hasattr(method, '_enum_key'):
|
|
149
|
+
method_info.update({
|
|
150
|
+
'type': "one",
|
|
151
|
+
'stage': method._stage,
|
|
152
|
+
'key': method._enum_key,
|
|
153
|
+
'pass_to': method._pass_to,
|
|
154
|
+
'parallel': method._parallel,
|
|
155
|
+
})
|
|
156
|
+
elif not hasattr(method, '_stage'):
|
|
157
|
+
# 全阶段方法但没有指定特定阶段
|
|
158
|
+
method_info['type'] = "all"
|
|
159
|
+
else:
|
|
160
|
+
# 特定阶段的全量处理方法
|
|
161
|
+
method_info.update({
|
|
162
|
+
'type': "all",
|
|
163
|
+
'stage': method._stage,
|
|
164
|
+
})
|
|
165
|
+
|
|
166
|
+
run_list[process_name] = method_info
|
|
167
|
+
|
|
168
|
+
return run_list
|
|
169
|
+
|
|
170
|
+
def _execute_process(self, board, process_info):
|
|
171
|
+
"""
|
|
172
|
+
执行单个处理过程
|
|
173
|
+
|
|
174
|
+
Args:
|
|
175
|
+
board: 执行面板对象
|
|
176
|
+
process_info (dict): 处理方法的元数据信息
|
|
177
|
+
|
|
178
|
+
根据处理类型执行不同逻辑:
|
|
179
|
+
- "all"类型:直接调用方法处理整个数据面板
|
|
180
|
+
- "one"类型:遍历数据项,逐个处理
|
|
181
|
+
"""
|
|
182
|
+
if process_info['type'] == "all":
|
|
183
|
+
# 全量处理,直接传入整个面板
|
|
184
|
+
process_info['func'](board)
|
|
185
|
+
elif process_info['type'] == "one":
|
|
186
|
+
# 单项处理
|
|
187
|
+
if len(board.current_data) == 0:
|
|
188
|
+
board.logs.append("当前数据为空,跳过单项处理。")
|
|
189
|
+
return
|
|
190
|
+
|
|
191
|
+
# 获取要处理的数据项
|
|
192
|
+
if isinstance(board.current_data, list):
|
|
193
|
+
enum_datas = board.current_data
|
|
194
|
+
else:
|
|
195
|
+
enum_datas = board.current_data[process_info['key']]
|
|
196
|
+
|
|
197
|
+
# 从指定位置开始处理
|
|
198
|
+
enum_datas = enum_datas[process_info["pass_to"]:]
|
|
199
|
+
|
|
200
|
+
if process_info['parallel']:
|
|
201
|
+
# 并行处理
|
|
202
|
+
def do_func(item):
|
|
203
|
+
return process_info['func'](item, board)
|
|
204
|
+
|
|
205
|
+
# 使用多线程处理并显示进度条
|
|
206
|
+
for item, result in board.process_bar.iter_bar(
|
|
207
|
+
execute_multitask(enum_datas, do_func,
|
|
208
|
+
thread_count=self.multi_thread_count,
|
|
209
|
+
max_queue_buffer=self.multi_thread_count * 10),
|
|
210
|
+
key=process_info['name'], max_count=len(enum_datas)):
|
|
211
|
+
if result is not None:
|
|
212
|
+
# 保存处理结果(只对支持属性设置的对象)
|
|
213
|
+
try:
|
|
214
|
+
if not hasattr(item, 'process_result'):
|
|
215
|
+
item.process_result = {}
|
|
216
|
+
item.process_result[process_info['name']] = result
|
|
217
|
+
except (AttributeError, TypeError):
|
|
218
|
+
# 对于基本类型(如 int, str),无法设置属性,忽略结果保存
|
|
219
|
+
pass
|
|
220
|
+
else:
|
|
221
|
+
# 串行处理
|
|
222
|
+
for item in board.process_bar.iter_bar(enum_datas, key=process_info['name']):
|
|
223
|
+
result = process_info['func'](item, board)
|
|
224
|
+
if result is not None:
|
|
225
|
+
# 保存处理结果(只对支持属性设置的对象)
|
|
226
|
+
try:
|
|
227
|
+
if not hasattr(item, 'process_result'):
|
|
228
|
+
item.process_result = {}
|
|
229
|
+
item.process_result[process_info['name']] = result
|
|
230
|
+
except (AttributeError, TypeError):
|
|
231
|
+
# 对于基本类型(如 int, str),无法设置属性,忽略结果保存
|
|
232
|
+
pass
|
|
233
|
+
|
|
234
|
+
def _execute_buffer(self, board, process_info):
|
|
235
|
+
"""
|
|
236
|
+
执行带缓冲的处理过程
|
|
237
|
+
|
|
238
|
+
Args:
|
|
239
|
+
board: 执行面板对象
|
|
240
|
+
process_info (dict): 处理方法的元数据信息
|
|
241
|
+
|
|
242
|
+
Returns:
|
|
243
|
+
tuple: (是否使用了缓冲, 缓冲文件路径)
|
|
244
|
+
|
|
245
|
+
缓冲机制:
|
|
246
|
+
- 如果启用缓冲且该方法支持缓冲,会尝试从缓冲中恢复结果
|
|
247
|
+
- 如果缓冲不存在,执行处理并保存结果到缓冲
|
|
248
|
+
- 缓冲键值基于方法名生成
|
|
249
|
+
"""
|
|
250
|
+
if self.buffer_switch and process_info['buffer']:
|
|
251
|
+
buffer_key = process_info['name'].replace('.', '_')
|
|
252
|
+
|
|
253
|
+
if has_item_key(buffer_key):
|
|
254
|
+
# 从缓冲恢复数据
|
|
255
|
+
board.serialize_tuple = get_buffer_item(buffer_key)
|
|
256
|
+
return True, buffer_key
|
|
257
|
+
else:
|
|
258
|
+
# 执行处理并保存到缓冲
|
|
259
|
+
self._execute_process(board, process_info)
|
|
260
|
+
buffer_item(buffer_key, board.serialize_tuple)
|
|
261
|
+
flush()
|
|
262
|
+
return False, buffer_key
|
|
263
|
+
else:
|
|
264
|
+
# 不使用缓冲,直接执行
|
|
265
|
+
self._execute_process(board, process_info)
|
|
266
|
+
return False, ""
|
|
267
|
+
|
|
268
|
+
def _execute_stage(self, board: ExecutionBoard):
|
|
269
|
+
"""
|
|
270
|
+
执行当前阶段的所有处理过程
|
|
271
|
+
|
|
272
|
+
Args:
|
|
273
|
+
board: 执行面板对象,包含当前阶段信息
|
|
274
|
+
|
|
275
|
+
按优先级顺序执行当前阶段的所有处理方法,记录执行信息和耗时
|
|
276
|
+
"""
|
|
277
|
+
if board.current_stage in self.sub_processes:
|
|
278
|
+
stage_processes = self.sub_processes[board.current_stage]
|
|
279
|
+
|
|
280
|
+
for process_info in board.process_bar.iter_bar(stage_processes, key="sub_process"):
|
|
281
|
+
if not process_info['enable']:
|
|
282
|
+
board.logs.append(f"跳过已禁用的处理过程: {process_info['name']}")
|
|
283
|
+
continue
|
|
284
|
+
|
|
285
|
+
board.logs.append(
|
|
286
|
+
f"开始处理: {process_info['name']}, "
|
|
287
|
+
f"优先级: {process_info['priority']}, "
|
|
288
|
+
f"使用缓冲: {process_info['buffer']}"
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
start_timestamp = time.time()
|
|
292
|
+
use_buffer, buffer_file = self._execute_buffer(board, process_info)
|
|
293
|
+
|
|
294
|
+
# 记录处理信息
|
|
295
|
+
board.sub_process_info.append({
|
|
296
|
+
"process_name": process_info['name'],
|
|
297
|
+
"stage": board.current_stage,
|
|
298
|
+
"use_buffer": use_buffer,
|
|
299
|
+
"buffer_file": buffer_file,
|
|
300
|
+
"cost": time.time() - start_timestamp,
|
|
301
|
+
})
|
|
302
|
+
|
|
303
|
+
def __call__(self, **kwargs):
|
|
304
|
+
"""
|
|
305
|
+
执行整个处理管道
|
|
306
|
+
|
|
307
|
+
Args:
|
|
308
|
+
**kwargs: 传递给执行面板的初始参数
|
|
309
|
+
|
|
310
|
+
Returns:
|
|
311
|
+
ExecutionBoard: 执行完成后的面板对象,包含所有阶段的处理结果
|
|
312
|
+
|
|
313
|
+
示例:
|
|
314
|
+
```python
|
|
315
|
+
# 传入初始数据
|
|
316
|
+
result = pipeline(data=[1, 2, 3], config={"debug": True})
|
|
317
|
+
|
|
318
|
+
# 访问结果
|
|
319
|
+
print(result.logs) # 查看执行日志
|
|
320
|
+
print(result.sub_process_info) # 查看各步骤执行信息
|
|
321
|
+
print(result.stages) # 查看各阶段数据
|
|
322
|
+
```
|
|
323
|
+
"""
|
|
324
|
+
board = ExecutionBoard()
|
|
325
|
+
|
|
326
|
+
# 设置初始参数
|
|
327
|
+
for k, v in kwargs.items():
|
|
328
|
+
setattr(board, k, v)
|
|
329
|
+
|
|
330
|
+
# 按顺序执行各阶段
|
|
331
|
+
for stage in board.process_bar.iter_bar(self.all_stages, key='stages'):
|
|
332
|
+
board.set_stage(stage)
|
|
333
|
+
self._execute_stage(board)
|
|
334
|
+
|
|
335
|
+
return board
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
"""
|
|
339
|
+
改进建议和可增加的功能:
|
|
340
|
+
|
|
341
|
+
1. 错误处理改进:
|
|
342
|
+
- 添加异常捕获和重试机制
|
|
343
|
+
- 支持容错模式,某个步骤失败不影响后续执行
|
|
344
|
+
- 添加详细的错误日志和堆栈信息
|
|
345
|
+
|
|
346
|
+
2. 性能优化:
|
|
347
|
+
- 支持动态调整线程数
|
|
348
|
+
- 添加内存使用监控
|
|
349
|
+
- 支持大数据流式处理,避免内存溢出
|
|
350
|
+
|
|
351
|
+
3. 功能增强:
|
|
352
|
+
- 支持条件执行(根据前一阶段结果决定是否执行)
|
|
353
|
+
- 支持分支和合并逻辑
|
|
354
|
+
- 添加钩子函数(before/after hooks)
|
|
355
|
+
- 支持暂停和恢复执行
|
|
356
|
+
|
|
357
|
+
4. 监控和调试:
|
|
358
|
+
- 添加详细的性能指标收集
|
|
359
|
+
- 支持实时监控执行状态
|
|
360
|
+
- 添加可视化的执行流程图
|
|
361
|
+
- 支持断点调试模式
|
|
362
|
+
|
|
363
|
+
5. 配置管理:
|
|
364
|
+
- 支持从配置文件加载管道定义
|
|
365
|
+
- 支持动态修改处理参数
|
|
366
|
+
- 添加参数验证机制
|
|
367
|
+
|
|
368
|
+
6. 扩展性:
|
|
369
|
+
- 支持插件系统
|
|
370
|
+
- 添加事件驱动机制
|
|
371
|
+
- 支持分布式执行
|
|
372
|
+
|
|
373
|
+
示例扩展实现:
|
|
374
|
+
|
|
375
|
+
class AdvancedPipeline(Pipeline):
|
|
376
|
+
def __init__(self, *args, **kwargs):
|
|
377
|
+
super().__init__(*args, **kwargs)
|
|
378
|
+
self.error_handlers = {}
|
|
379
|
+
self.retry_config = {"max_retries": 3, "delay": 1}
|
|
380
|
+
|
|
381
|
+
def add_error_handler(self, stage, handler):
|
|
382
|
+
self.error_handlers[stage] = handler
|
|
383
|
+
|
|
384
|
+
def set_retry_config(self, max_retries=3, delay=1):
|
|
385
|
+
self.retry_config = {"max_retries": max_retries, "delay": delay}
|
|
386
|
+
"""
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tketool.pipeline
|
|
3
|
+
Version: 1.3.5
|
|
4
|
+
Summary: Stage-based processing pipeline and injection decorators for tketool
|
|
5
|
+
Author-email: Ke <jiangke1207@icloud.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://pypi.org/project/tketool.pipeline/
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Python: >=3.10
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: tketool.core[console]==1.3.5
|
|
16
|
+
Provides-Extra: test
|
|
17
|
+
Requires-Dist: pytest<9,>=8; extra == "test"
|
|
18
|
+
|
|
19
|
+
# tketool.pipeline
|
|
20
|
+
|
|
21
|
+
Small stage-based processing pipelines with ordered decorators, optional
|
|
22
|
+
parallel item processing, and local buffering.
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install tketool.pipeline
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
```python
|
|
29
|
+
from enum import Enum
|
|
30
|
+
|
|
31
|
+
from tketool.pipeline import Pipeline, invoke_all
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class Stage(Enum):
|
|
35
|
+
PREPARE = "prepare"
|
|
36
|
+
RUN = "run"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Steps:
|
|
40
|
+
@invoke_all(Stage.PREPARE.value, priority=10)
|
|
41
|
+
def prepare(self, board):
|
|
42
|
+
board.current_data = {"ready": True}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
pipeline = Pipeline(Stage, acts=[Steps()])
|
|
46
|
+
result = pipeline()
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
The retired `tketool.injections` path is not shipped.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
src/tketool.pipeline.egg-info/PKG-INFO
|
|
4
|
+
src/tketool.pipeline.egg-info/SOURCES.txt
|
|
5
|
+
src/tketool.pipeline.egg-info/dependency_links.txt
|
|
6
|
+
src/tketool.pipeline.egg-info/requires.txt
|
|
7
|
+
src/tketool.pipeline.egg-info/top_level.txt
|
|
8
|
+
src/tketool/pipeline/__init__.py
|
|
9
|
+
src/tketool/pipeline/execution.py
|
|
10
|
+
src/tketool/pipeline/pipeline.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
tketool
|