diffusers-workflow 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
  2. diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
  3. diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
  4. diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
  5. diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
  6. diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
  7. dw/__init__.py +440 -0
  8. dw/adapter_compatibility.py +226 -0
  9. dw/arguments.py +1231 -0
  10. dw/assessment_rules.py +159 -0
  11. dw/assets.py +130 -0
  12. dw/cache_blocks.json +16 -0
  13. dw/cache_blocks.py +146 -0
  14. dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
  15. dw/content_types.py +150 -0
  16. dw/dissolve_frame_errors.py +121 -0
  17. dw/docs/ACCELERATION.md +352 -0
  18. dw/docs/AGENT_LOOP.md +95 -0
  19. dw/docs/DEPENDENCIES.md +91 -0
  20. dw/docs/IP_ADAPTER.md +109 -0
  21. dw/docs/LORAS.md +131 -0
  22. dw/docs/MCP.md +517 -0
  23. dw/docs/PROMPT_WEIGHTING.md +78 -0
  24. dw/docs/QUANTIZATION.md +230 -0
  25. dw/docs/RECIPES_24GB.md +201 -0
  26. dw/docs/RELEASING.md +195 -0
  27. dw/docs/REMOTE.md +140 -0
  28. dw/docs/REPL_COMMANDS.md +121 -0
  29. dw/docs/REPL_WORKER_GUIDE.md +51 -0
  30. dw/docs/SECURITY.md +272 -0
  31. dw/docs/SECURITY_QUICKREF.md +112 -0
  32. dw/docs/SERVER.md +679 -0
  33. dw/docs/TASKS.md +1741 -0
  34. dw/docs/TESTING.md +71 -0
  35. dw/docs/WORKFLOW_GUIDE.md +2038 -0
  36. dw/docs/WORKSPACES.md +316 -0
  37. dw/download_watch.py +335 -0
  38. dw/elision.py +306 -0
  39. dw/events.py +275 -0
  40. dw/for_each.py +409 -0
  41. dw/host_memory.py +258 -0
  42. dw/host_memory_projection.py +230 -0
  43. dw/hub_cache.py +432 -0
  44. dw/introspection.py +1228 -0
  45. dw/kernel_availability.py +208 -0
  46. dw/locations.py +599 -0
  47. dw/log_setup.py +45 -0
  48. dw/loudness.py +82 -0
  49. dw/media_audio.py +217 -0
  50. dw/media_frames.py +367 -0
  51. dw/media_info.py +297 -0
  52. dw/pipeline_processors/chain.py +821 -0
  53. dw/pipeline_processors/config_objects.py +237 -0
  54. dw/pipeline_processors/pipeline.py +2297 -0
  55. dw/pipeline_processors/remote.py +46 -0
  56. dw/plan.py +920 -0
  57. dw/previous_results.py +411 -0
  58. dw/probe_paths.py +59 -0
  59. dw/prompt_schema.json +48 -0
  60. dw/prompt_weighting.py +378 -0
  61. dw/prompts.py +159 -0
  62. dw/realize.py +250 -0
  63. dw/reference_limits.py +215 -0
  64. dw/reference_names.py +125 -0
  65. dw/repl.py +338 -0
  66. dw/repl_commands.py +836 -0
  67. dw/repl_worker.py +159 -0
  68. dw/result.py +1720 -0
  69. dw/result_fps.py +82 -0
  70. dw/run.py +162 -0
  71. dw/runs.py +768 -0
  72. dw/scalar_result_validation.py +97 -0
  73. dw/schema.py +283 -0
  74. dw/security.py +1038 -0
  75. dw/select_validation.py +115 -0
  76. dw/serve.py +277 -0
  77. dw/server/__init__.py +2 -0
  78. dw/server/app.py +4586 -0
  79. dw/server/assess.py +132 -0
  80. dw/server/catalog_shape.py +487 -0
  81. dw/server/enhancers.py +129 -0
  82. dw/server/exports.py +480 -0
  83. dw/server/guides.py +257 -0
  84. dw/server/jobs.py +1561 -0
  85. dw/server/mcp_mount.py +95 -0
  86. dw/server/netinfo.py +124 -0
  87. dw/server/observed_cost.py +379 -0
  88. dw/server/sysinfo.py +71 -0
  89. dw/server/ui/assets/abap-08VXUWAP.js +1 -0
  90. dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
  91. dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
  92. dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
  93. dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
  94. dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
  95. dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
  96. dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
  97. dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
  98. dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
  99. dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
  100. dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
  101. dw/server/ui/assets/css-DIMkf-bt.js +3 -0
  102. dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
  103. dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
  104. dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
  105. dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
  106. dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
  107. dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
  108. dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
  109. dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
  110. dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
  111. dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
  112. dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
  113. dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
  114. dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
  115. dw/server/ui/assets/go-C-y9NEjX.js +1 -0
  116. dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
  117. dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
  118. dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
  119. dw/server/ui/assets/html-PhsdjHSr.js +1 -0
  120. dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
  121. dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
  122. dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
  123. dw/server/ui/assets/index-DgrYhQd9.js +43 -0
  124. dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
  125. dw/server/ui/assets/java-BEtHBSE6.js +1 -0
  126. dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
  127. dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
  128. dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
  129. dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
  130. dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
  131. dw/server/ui/assets/less-B9JPFI3C.js +2 -0
  132. dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
  133. dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
  134. dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
  135. dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
  136. dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
  137. dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
  138. dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
  139. dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
  140. dw/server/ui/assets/monaco--ixms01u.css +1 -0
  141. dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
  142. dw/server/ui/assets/msdax-DauUninz.js +1 -0
  143. dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
  144. dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
  145. dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
  146. dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
  147. dw/server/ui/assets/perl-oz_6vUea.js +1 -0
  148. dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
  149. dw/server/ui/assets/php-nr791fC2.js +1 -0
  150. dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
  151. dw/server/ui/assets/postiats-43DmfD33.js +1 -0
  152. dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
  153. dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
  154. dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
  155. dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
  156. dw/server/ui/assets/python-Bcn70HdC.js +1 -0
  157. dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
  158. dw/server/ui/assets/r-BwWrilGY.js +1 -0
  159. dw/server/ui/assets/razor-D1HmNnby.js +1 -0
  160. dw/server/ui/assets/redis-ClamHrr6.js +1 -0
  161. dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
  162. dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
  163. dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
  164. dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
  165. dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
  166. dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
  167. dw/server/ui/assets/scheme-BeGwcela.js +1 -0
  168. dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
  169. dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
  170. dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
  171. dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
  172. dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
  173. dw/server/ui/assets/sql-NEE52Syq.js +1 -0
  174. dw/server/ui/assets/st-DbInun42.js +1 -0
  175. dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
  176. dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
  177. dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
  178. dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
  179. dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
  180. dw/server/ui/assets/twig-De2hgUGE.js +1 -0
  181. dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
  182. dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
  183. dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
  184. dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
  185. dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
  186. dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
  187. dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
  188. dw/server/ui/index.html +17 -0
  189. dw/server/updater.py +192 -0
  190. dw/settings.py +98 -0
  191. dw/shot_span_preflight.py +116 -0
  192. dw/shots.py +359 -0
  193. dw/slice_preflight.py +148 -0
  194. dw/step.py +187 -0
  195. dw/step_cache.py +442 -0
  196. dw/subfolders.py +107 -0
  197. dw/task_domains.py +307 -0
  198. dw/tasks/assess.py +826 -0
  199. dw/tasks/audio_transcription.py +88 -0
  200. dw/tasks/audio_utils.py +1862 -0
  201. dw/tasks/background_remover.py +43 -0
  202. dw/tasks/borders.py +113 -0
  203. dw/tasks/compose_text.py +74 -0
  204. dw/tasks/concat_videos.py +300 -0
  205. dw/tasks/depth_estimator.py +54 -0
  206. dw/tasks/diffusion_upscale.py +109 -0
  207. dw/tasks/dissolve_videos.py +342 -0
  208. dw/tasks/format_messages.py +24 -0
  209. dw/tasks/gather.py +173 -0
  210. dw/tasks/grade.py +97 -0
  211. dw/tasks/image_to_text.py +43 -0
  212. dw/tasks/image_utils.py +764 -0
  213. dw/tasks/interpolate_frames.py +252 -0
  214. dw/tasks/judge.py +68 -0
  215. dw/tasks/model_cache.py +55 -0
  216. dw/tasks/pair_audio.py +268 -0
  217. dw/tasks/qr_code.py +19 -0
  218. dw/tasks/restore_faces.py +175 -0
  219. dw/tasks/rife_model.py +192 -0
  220. dw/tasks/segment.py +121 -0
  221. dw/tasks/select.py +111 -0
  222. dw/tasks/speech_generation.py +228 -0
  223. dw/tasks/stabilize.py +129 -0
  224. dw/tasks/task.py +920 -0
  225. dw/tasks/tensor_image.py +57 -0
  226. dw/tasks/text_generation.py +169 -0
  227. dw/tasks/text_sections.py +80 -0
  228. dw/tasks/upscale.py +203 -0
  229. dw/tasks/video_utils.py +624 -0
  230. dw/tasks/zoe_depth.py +71 -0
  231. dw/teacache.py +381 -0
  232. dw/teacache_models.json +99 -0
  233. dw/test.py +29 -0
  234. dw/type_helpers.py +231 -0
  235. dw/validate.py +68 -0
  236. dw/variable_constraints.py +444 -0
  237. dw/variables.py +443 -0
  238. dw/video_extensions.py +141 -0
  239. dw/vram_estimate.py +116 -0
  240. dw/worker.py +764 -0
  241. dw/workflow.py +2007 -0
  242. dw/workflow_schema.json +1346 -0
  243. dw/workflow_sources.py +383 -0
  244. dw/workflows/h3_context_ir.json +57 -0
  245. dw/workflows/test.json +31 -0
  246. dw/workspace.py +730 -0
  247. dw_mcp/__init__.py +6 -0
  248. dw_mcp/__main__.py +133 -0
  249. dw_mcp/assets.py +336 -0
  250. dw_mcp/authoring.py +114 -0
  251. dw_mcp/catalog.py +360 -0
  252. dw_mcp/client.py +486 -0
  253. dw_mcp/diagnose.py +371 -0
  254. dw_mcp/exports.py +84 -0
  255. dw_mcp/guides.py +35 -0
  256. dw_mcp/media.py +638 -0
  257. dw_mcp/models.py +97 -0
  258. dw_mcp/prompts.py +104 -0
  259. dw_mcp/server.py +1343 -0
  260. dw_mcp/workspaces.py +212 -0
dw/server/jobs.py ADDED
@@ -0,0 +1,1561 @@
1
+ """Job queue over the persistent worker process.
2
+
3
+ One runner thread executes jobs FIFO against the single GPU worker - the
4
+ same WorkerManager the REPL uses. Jobs collect their progress events with
5
+ sequence numbers so an SSE client can attach late (or reconnect) and replay
6
+ from where it left off.
7
+ """
8
+
9
+ import os
10
+ import copy
11
+ import json
12
+ import queue
13
+ import secrets
14
+ import sqlite3
15
+ import time
16
+ import uuid
17
+ import logging
18
+ import threading
19
+
20
+ from ..download_watch import format_progress
21
+ from ..repl_worker import WorkerManager
22
+ from ..workflow import SEED_BITS, workflow_from_file, workflow_from_definition
23
+ from ..introspection import workflow_argument_warnings
24
+ from ..schema import format_validation_errors
25
+ from ..variables import argument_errors
26
+ from ..security import (
27
+ SecurityError,
28
+ validate_json_size,
29
+ validate_output_path,
30
+ validate_path,
31
+ validate_workflow_path,
32
+ )
33
+ from ..realize import VARIABLE_PREFIX
34
+ from ..runs import REALIZED_FILE_NAME
35
+ from ..settings import resolve_path
36
+ from ..workspace import DEFAULT_WORKSPACE_NAME
37
+ from .observed_cost import EVENT_CAP, LOADING_MARKER
38
+
39
+ logger = logging.getLogger("dw")
40
+
41
+ QUEUED = "queued"
42
+ RUNNING = "running"
43
+ SUCCEEDED = "succeeded"
44
+ FAILED = "failed"
45
+ CANCELLED = "cancelled"
46
+ TERMINAL_STATES = (SUCCEEDED, FAILED, CANCELLED)
47
+
48
+ # Which form of cost acknowledgement a job was queued with (#85): none (the
49
+ # web UI and every HTTP caller that sends nothing), a bare boolean, or one
50
+ # bound to the plan that was validated
51
+ ACK_NONE = "none"
52
+ ACK_BOOLEAN = "boolean"
53
+ ACK_BOUND = "bound"
54
+
55
+ # The spec fields a rerun needs - shared by persistence and live rerun
56
+ RERUN_SPEC_KEYS = (
57
+ "workflow_path",
58
+ "workflow",
59
+ "base_dir",
60
+ # A rerun belongs in the workspace the original ran in, so the roots
61
+ # that decided that are part of what history keeps
62
+ "workspace",
63
+ "output_dir",
64
+ "asset_dir",
65
+ # so a rerun is attributed to the same catalog entry
66
+ "catalog_name",
67
+ "workflow_dir",
68
+ # what the original run was consented to, kept for the record - a
69
+ # rerun's own request decides its form
70
+ "acknowledged_cost",
71
+ )
72
+
73
+ # Finished jobs kept in memory for SSE replay grace; older ones live in
74
+ # history only, so a long-running server's memory stays bounded
75
+ TERMINAL_JOBS_KEPT = 20
76
+
77
+ # A long run emits thousands of progress events; the tail is what explains
78
+ # the outcome. Bounded so history stays a summary store, not an event log
79
+ MAX_PERSISTED_EVENTS = 200
80
+
81
+
82
+ class JobHistory:
83
+ """Finished jobs, persisted so the Jobs view survives server restarts.
84
+
85
+ Records land at terminal state only - a crash mid-run loses that run's
86
+ row, which is the right trade for never blocking the runner on disk.
87
+ The last MAX_PERSISTED_EVENTS progress events ride along, so a job can
88
+ still explain itself after a restart; everything earlier is dropped.
89
+ """
90
+
91
+ def __init__(self, db_path):
92
+ self.db_path = str(db_path)
93
+ self._lock = threading.Lock()
94
+ with self._connect() as connection:
95
+ connection.execute("""CREATE TABLE IF NOT EXISTS jobs (
96
+ id TEXT PRIMARY KEY,
97
+ workflow TEXT,
98
+ status TEXT,
99
+ created_at REAL,
100
+ started_at REAL,
101
+ finished_at REAL,
102
+ arguments TEXT,
103
+ spec TEXT,
104
+ manifest TEXT,
105
+ warnings TEXT,
106
+ error TEXT,
107
+ events TEXT
108
+ )""")
109
+ # Databases written before events were persisted are missing the
110
+ # column; ALTER is the whole migration, and rows keep NULL
111
+ columns = {row[1] for row in connection.execute("PRAGMA table_info(jobs)")}
112
+ if "events" not in columns:
113
+ connection.execute("ALTER TABLE jobs ADD COLUMN events TEXT")
114
+ # Every row predating workspaces belongs to the default one -
115
+ # history that cannot say which workspace a job ran in stops
116
+ # making sense the moment there are two
117
+ if "workspace" not in columns:
118
+ connection.execute(
119
+ "ALTER TABLE jobs ADD COLUMN workspace TEXT DEFAULT 'default'"
120
+ )
121
+ connection.execute(
122
+ "UPDATE jobs SET workspace = 'default' WHERE workspace IS NULL"
123
+ )
124
+ # The catalog name the job was run from, beside `workflow` (the
125
+ # definition's id). Ids are not unique across a catalog forever;
126
+ # names are, and a later runtime-by-workflow join wants the exact
127
+ # one. Rows before this column stay NULL: old history is
128
+ # unjoinable, new history is exact
129
+ if "workflow_name" not in columns:
130
+ connection.execute("ALTER TABLE jobs ADD COLUMN workflow_name TEXT")
131
+ # Which run of the workflow this job was - the directory under the
132
+ # output root that holds its manifest and its realized workflow.
133
+ # NULL for every row predating run tracking, and the manager
134
+ # refuses to guess one from file paths
135
+ if "run_id" not in columns:
136
+ connection.execute("ALTER TABLE jobs ADD COLUMN run_id TEXT")
137
+ if "run_dir" not in columns:
138
+ connection.execute("ALTER TABLE jobs ADD COLUMN run_dir TEXT")
139
+ # That run's ordinal among the workflow's runs - the 'v4' the
140
+ # gallery shows. NULL before the column, and for a job that
141
+ # never opened a run
142
+ if "run_version" not in columns:
143
+ connection.execute("ALTER TABLE jobs ADD COLUMN run_version INTEGER")
144
+ # Which form of cost acknowledgement queued the job. Rows before
145
+ # the column are 'none' - nothing recorded is nothing recorded
146
+ if "acknowledged" not in columns:
147
+ connection.execute(
148
+ "ALTER TABLE jobs ADD COLUMN acknowledged TEXT DEFAULT 'none'"
149
+ )
150
+ # The worker's own high-water mark for this run (#243) - NULL for
151
+ # a row predating the column and for any run that never reported
152
+ # one (cancelled/errored before the worker's final memory_info)
153
+ if "host_memory_peak_rss_mb" not in columns:
154
+ connection.execute(
155
+ "ALTER TABLE jobs ADD COLUMN host_memory_peak_rss_mb REAL"
156
+ )
157
+ # This job's own contribution to that process-lifetime figure -
158
+ # growth since the job's first phase-boundary reading, or its
159
+ # current rss when it caused no growth (#272). NULL for a row
160
+ # predating the column and for any run that never got a
161
+ # memory_info message at all
162
+ if "host_memory_job_peak_rss_mb" not in columns:
163
+ connection.execute(
164
+ "ALTER TABLE jobs ADD COLUMN host_memory_job_peak_rss_mb REAL"
165
+ )
166
+
167
+ def _connect(self):
168
+ # WAL mode lets a reader (the web UI polling job status, an MCP
169
+ # get_job call) proceed without blocking behind whatever write the
170
+ # worker is mid-transaction on, and vice versa - the default
171
+ # rollback-journal mode takes a database-wide lock for the
172
+ # duration of a write. journal_mode is a property of the database
173
+ # file, not the connection, but PRAGMA is cheap and idempotent, so
174
+ # it is set on every connect rather than assumed to have stuck.
175
+ connection = sqlite3.connect(self.db_path, timeout=5)
176
+ connection.execute("PRAGMA journal_mode=WAL")
177
+ return connection
178
+
179
+ def record(self, job):
180
+ # The spec's workflow_name/warnings are derived; keep what rerun needs
181
+ rerun_spec = {key: job.spec[key] for key in RERUN_SPEC_KEYS if key in job.spec}
182
+ with self._lock, self._connect() as connection:
183
+ connection.execute(
184
+ "INSERT OR REPLACE INTO jobs (id, workflow, status, created_at,"
185
+ " started_at, finished_at, arguments, spec, manifest, warnings,"
186
+ " error, events, workspace, workflow_name, run_id, run_dir,"
187
+ " acknowledged, host_memory_peak_rss_mb,"
188
+ " host_memory_job_peak_rss_mb, run_version) VALUES"
189
+ " (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
190
+ (
191
+ job.id,
192
+ job.workflow_name,
193
+ job.status,
194
+ job.created_at,
195
+ job.started_at,
196
+ job.finished_at,
197
+ json.dumps(job.spec.get("arguments", {}), default=str),
198
+ json.dumps(rerun_spec, default=str),
199
+ json.dumps(job.manifest, default=str),
200
+ json.dumps(job.warnings, default=str),
201
+ job.error,
202
+ json.dumps(job.events[-MAX_PERSISTED_EVENTS:], default=str),
203
+ job.spec.get("workspace") or DEFAULT_WORKSPACE_NAME,
204
+ job.catalog_name,
205
+ job.run_id,
206
+ job.run_dir,
207
+ job.acknowledged,
208
+ # A test double or an older in-memory Job predating this
209
+ # column reports None here rather than failing record()
210
+ # (#243) - the same "absent means unknown" the column
211
+ # itself allows
212
+ getattr(job, "host_memory_peak_rss_mb", None),
213
+ getattr(job, "host_memory_job_peak_rss_mb", None),
214
+ getattr(job, "run_version", None),
215
+ ),
216
+ )
217
+
218
+ def recent_summaries(self, limit=200, workspace=None, statuses=None):
219
+ """Summary rows only - the jobs list is polled, and parsing four JSON
220
+ blobs per row just to show six scalars was pure waste.
221
+
222
+ `workspace` filters to one workspace's rows; omitted, history spans
223
+ all of them the way the list already did before workspaces existed.
224
+ `statuses` filters to a set of terminal states - in SQL rather than
225
+ over the returned rows, or the newest-first cap above would be
226
+ spent on rows the filter then drops.
227
+ """
228
+ query = (
229
+ "SELECT id, workflow, status, created_at, started_at, finished_at,"
230
+ " workspace, workflow_name, run_id, acknowledged, run_version"
231
+ " FROM jobs"
232
+ )
233
+ params = []
234
+ clauses = []
235
+ if workspace:
236
+ clauses.append("workspace = ?")
237
+ params.append(workspace)
238
+ if statuses:
239
+ statuses = list(statuses)
240
+ placeholders = ", ".join("?" for _ in statuses)
241
+ clauses.append(f"status IN ({placeholders})")
242
+ params.extend(statuses)
243
+ if clauses:
244
+ query += " WHERE " + " AND ".join(clauses)
245
+ query += " ORDER BY created_at DESC LIMIT ?"
246
+ params.append(limit)
247
+ with self._lock, self._connect() as connection:
248
+ rows = connection.execute(query, params).fetchall()
249
+ return [
250
+ {
251
+ "id": row[0],
252
+ "workflow": row[1],
253
+ "status": row[2],
254
+ "created_at": row[3],
255
+ "started_at": row[4],
256
+ "finished_at": row[5],
257
+ "workspace": row[6] or DEFAULT_WORKSPACE_NAME,
258
+ "workflow_name": row[7],
259
+ "run_id": row[8],
260
+ "acknowledged": row[9] or ACK_NONE,
261
+ "run_version": row[10],
262
+ "historical": True,
263
+ }
264
+ for row in rows
265
+ ]
266
+
267
+ def get(self, job_id):
268
+ with self._lock, self._connect() as connection:
269
+ row = connection.execute(
270
+ "SELECT id, workflow, status, created_at, started_at, finished_at,"
271
+ " arguments, spec, manifest, warnings, error, workspace,"
272
+ " workflow_name, run_id, run_dir, acknowledged, events,"
273
+ " run_version FROM jobs WHERE id = ?",
274
+ (job_id,),
275
+ ).fetchone()
276
+ return self._to_detail(row) if row else None
277
+
278
+ def watermark(self):
279
+ """How far the table has got - what a derived figure caches against.
280
+
281
+ A job landing changes every observed cost and changes no file, so an
282
+ mtime cache cannot see it (dw/server/observed_cost.py). Counted over
283
+ `workflow_name IS NOT NULL` rather than every row, because
284
+ `orphan_workflow_history` (#274) detaches a deleted workflow's rows by
285
+ clearing that column rather than deleting the row - an ordinary
286
+ `COUNT(*)` would not move, and `ObservedCosts` would keep serving the
287
+ purged figure until an unrelated job happened to land. Counting only
288
+ the joinable rows falls by exactly the amount a purge detaches, the
289
+ same as a prune lowering it.
290
+ """
291
+ with self._lock, self._connect() as connection:
292
+ row = connection.execute(
293
+ "SELECT COUNT(*), MAX(finished_at) FROM jobs"
294
+ " WHERE workflow_name IS NOT NULL"
295
+ ).fetchone()
296
+ return (row[0], row[1]) if row else (0, None)
297
+
298
+ def finished_runs(self):
299
+ """Every successful, named run grouped by (workspace, workflow name),
300
+ as the rows an observed cost is derived from.
301
+
302
+ One query for the whole catalog rather than one per workflow. The
303
+ cold/warm split is decided in SQL on the persisted event tail - a
304
+ `loading` phase as `json.dumps` wrote it - so 200 events per row are
305
+ never parsed to answer a yes/no question, and whether that tail hit
306
+ its cap comes back too, because a run whose `loading` phase was
307
+ trimmed away has to count as neither rather than as warm.
308
+
309
+ Rows with no `workflow_name` (recorded before the column existed, or
310
+ run from an inline definition, or orphaned by `orphan_workflow_history`)
311
+ are unjoinable and left out. The workspace dimension is always in the
312
+ key here; whether a caller treats two workspaces as one history (a
313
+ shared catalog source, #154) or as separate (a workspace's own
314
+ writable copy, #274) is decided in `ObservedCosts.rows_for`, which is
315
+ the layer that knows which kind of source it was asked about.
316
+ """
317
+ with self._lock, self._connect() as connection:
318
+ rows = connection.execute(
319
+ "SELECT workflow_name, workspace, started_at, finished_at,"
320
+ " arguments, manifest, INSTR(COALESCE(events, ''), ?) > 0,"
321
+ " COALESCE(json_array_length(COALESCE(events, '[]')), 0) >= ?,"
322
+ " host_memory_peak_rss_mb, host_memory_job_peak_rss_mb"
323
+ " FROM jobs WHERE status = ? AND workflow_name IS NOT NULL"
324
+ " AND started_at IS NOT NULL AND finished_at IS NOT NULL",
325
+ (LOADING_MARKER, EVENT_CAP, SUCCEEDED),
326
+ ).fetchall()
327
+ grouped = {}
328
+ for (
329
+ name,
330
+ workspace,
331
+ started,
332
+ finished,
333
+ arguments,
334
+ manifest,
335
+ had_load,
336
+ at_cap,
337
+ peak_rss_mb,
338
+ job_peak_rss_mb,
339
+ ) in rows:
340
+ key = (workspace or DEFAULT_WORKSPACE_NAME, name)
341
+ grouped.setdefault(key, []).append(
342
+ {
343
+ "started_at": started,
344
+ "finished_at": finished,
345
+ "duration": finished - started,
346
+ "arguments": arguments,
347
+ "manifest": manifest,
348
+ "had_load": bool(had_load),
349
+ "events_at_cap": bool(at_cap),
350
+ "host_memory_peak_rss_mb": peak_rss_mb,
351
+ "host_memory_job_peak_rss_mb": job_peak_rss_mb,
352
+ }
353
+ )
354
+ return grouped
355
+
356
+ def orphan_workflow_history(self, workspace, workflow_name):
357
+ """Detach this (workspace, workflow_name)'s finished runs from cost
358
+ history (#274).
359
+
360
+ Deleting a workflow does not delete the job rows that ran it - those
361
+ stay for `list_jobs`/`get_job` and any other audit trail - but a name
362
+ reused afterwards, in this workspace or a fresh one copied from it,
363
+ must not inherit the old identity's figures. Setting `workflow_name`
364
+ to NULL is enough: `finished_runs()` already excludes rows where it
365
+ is NULL, the same rule that already excludes a run from an inline
366
+ definition.
367
+ """
368
+ with self._lock, self._connect() as connection:
369
+ connection.execute(
370
+ "UPDATE jobs SET workflow_name = NULL"
371
+ " WHERE workspace = ? AND workflow_name = ?",
372
+ (workspace, workflow_name),
373
+ )
374
+
375
+ def events_for(self, job_id):
376
+ """A finished job's persisted event tail. [] for a job recorded
377
+ before events were kept, None for a job history has never seen -
378
+ the caller needs to tell 'no events' from 'no such job'."""
379
+ with self._lock, self._connect() as connection:
380
+ row = connection.execute(
381
+ "SELECT events FROM jobs WHERE id = ?", (job_id,)
382
+ ).fetchone()
383
+ if row is None:
384
+ return None
385
+ if not row[0]:
386
+ return []
387
+ try:
388
+ return json.loads(row[0])
389
+ except json.JSONDecodeError:
390
+ return []
391
+
392
+ def job_for_file(self, file_name, workspace=None):
393
+ """The most recent job that actually wrote this output file.
394
+
395
+ LIKE metacharacters are escaped - generated names routinely contain
396
+ '_', which would otherwise match any character and let a similarly
397
+ named later job claim the file.
398
+
399
+ A manifest entry marked 'reused' is a step-cache hit republishing an
400
+ earlier run's files, so it is skipped: attribution belongs to the job
401
+ that wrote the file, not to every later run that reused it.
402
+
403
+ `workspace` narrows the scan to one workspace - two workspaces can
404
+ each produce a file with the same relative name, and without this a
405
+ later job in another workspace could wrongly claim the match.
406
+ """
407
+ escaped = (
408
+ file_name.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_")
409
+ )
410
+ # Unbounded on purpose: every later fixed-seed rerun republishes the
411
+ # file with 'reused', so a LIMIT would let the writing job fall out of
412
+ # the window after that many reruns and leave the file unattributed.
413
+ # The LIKE filter already restricts the scan to manifests naming it.
414
+ query = (
415
+ "SELECT id, status, manifest FROM jobs WHERE manifest LIKE ? ESCAPE '\\'"
416
+ )
417
+ params = [f"%{escaped}%"]
418
+ if workspace:
419
+ query += " AND workspace = ?"
420
+ params.append(workspace)
421
+ query += " ORDER BY finished_at DESC"
422
+ with self._lock, self._connect() as connection:
423
+ rows = connection.execute(query, params).fetchall()
424
+ for row in rows:
425
+ if self._manifest_wrote(row[2], file_name):
426
+ return {"id": row[0], "status": row[1]}
427
+ return None
428
+
429
+ @staticmethod
430
+ def _manifest_wrote(manifest_text, file_name):
431
+ """Whether this manifest names the file in an entry it wrote itself.
432
+
433
+ A manifest that will not parse falls back to the LIKE match that
434
+ found it - a row recorded before entries carried 'reused' cannot
435
+ have been a reuse anyway.
436
+ """
437
+ try:
438
+ manifest = json.loads(manifest_text)
439
+ except (TypeError, ValueError):
440
+ return True
441
+ if not isinstance(manifest, list):
442
+ return True
443
+ # A manifest entry names a file the way the run recorded it - a
444
+ # server-recorded manifest holds names relative to the output
445
+ # directory (_relative_output_names), a directly-run workflow's holds
446
+ # absolute paths. The caller names it relative to the output
447
+ # directory, so match on the tail either way - the same relationship
448
+ # the LIKE substring match relied on
449
+ wanted = file_name.replace(os.sep, "/")
450
+
451
+ def names_file(path):
452
+ normalized = path.replace(os.sep, "/")
453
+ return normalized == wanted or normalized.endswith("/" + wanted)
454
+
455
+ return any(
456
+ not entry.get("reused")
457
+ and any(names_file(path) for path in entry.get("files") or [])
458
+ for entry in manifest
459
+ if isinstance(entry, dict)
460
+ )
461
+
462
+ @staticmethod
463
+ def _to_detail(row):
464
+ def parse(text, fallback):
465
+ try:
466
+ return json.loads(text)
467
+ except (TypeError, ValueError):
468
+ return fallback
469
+
470
+ spec = parse(row[7], {})
471
+ # The persisted tail is capped at MAX_PERSISTED_EVENTS, and
472
+ # get_job_events serves that same tail - so counting it, rather than
473
+ # hardcoding 0, keeps event_count truthful about what a caller who
474
+ # pages through get_job_events will actually see (#289)
475
+ events = parse(row[16], [])
476
+ return {
477
+ "id": row[0],
478
+ "workflow": row[1],
479
+ "status": row[2],
480
+ "created_at": row[3],
481
+ "started_at": row[4],
482
+ "finished_at": row[5],
483
+ "arguments": parse(row[6], {}),
484
+ "spec": spec,
485
+ "manifest": parse(row[8], []),
486
+ "warnings": parse(row[9], []),
487
+ "error": row[10],
488
+ "workspace": row[11] or DEFAULT_WORKSPACE_NAME,
489
+ "workflow_name": row[12],
490
+ "run_id": row[13],
491
+ "run_dir": row[14],
492
+ "run_version": row[17],
493
+ "acknowledged": row[15] or ACK_NONE,
494
+ "acknowledged_cost": (spec or {}).get("acknowledged_cost"),
495
+ "traceback": None,
496
+ "event_count": len(events) if isinstance(events, list) else 0,
497
+ "historical": True,
498
+ }
499
+
500
+
501
+ class Job:
502
+ """One workflow execution request and everything observed about it."""
503
+
504
+ def __init__(self, spec):
505
+ self.id = uuid.uuid4().hex[:12]
506
+ self.spec = spec
507
+ self.workflow_name = spec.get("workflow_name", "unknown")
508
+ self.catalog_name = spec.get("catalog_name")
509
+ self.status = QUEUED
510
+ self.created_at = time.time()
511
+ self.started_at = None
512
+ self.finished_at = None
513
+ self.manifest = []
514
+ # A copy: run-time warnings are appended to this list (see
515
+ # _note_progress) and the spec is what a rerun is built from
516
+ self.warnings = list(spec.get("warnings", []))
517
+ self.error = None
518
+ self.traceback = None
519
+ # Which run this job turned out to be - reported by the worker's
520
+ # run_start event, unknown until then and forever for a job that
521
+ # never got that far
522
+ self.run_id = None
523
+ self.run_dir = None
524
+ self.run_version = None
525
+ # Which form of cost acknowledgement queued this job (#85)
526
+ self.acknowledged = spec.get("acknowledged") or ACK_NONE
527
+ # The worker's own high-water mark for this run, from its final
528
+ # memory_info message - None for a run that never got that far (#243)
529
+ self.host_memory_peak_rss_mb = None
530
+ # This job's own contribution to that process-lifetime figure -
531
+ # growth since the job's first phase boundary, or the job's current
532
+ # rss when it caused no growth (#272). None for a run that never got
533
+ # a memory_info message at all
534
+ self.host_memory_job_peak_rss_mb = None
535
+ self.events = []
536
+ # The running summary a poll reads - see _note_progress. Kept as the
537
+ # events arrive rather than derived from the log on request, because
538
+ # the log is trimmed to its last MAX_PERSISTED_EVENTS and a caller
539
+ # polling a long render should not have to page through it to learn
540
+ # that something moved
541
+ self.last_event_at = None
542
+ self.phase = None
543
+ self.phase_detail = None
544
+ self.phase_started_at = None
545
+ self.step_name = None
546
+ self.parent_step = None
547
+ self.step_index = None
548
+ self.total_steps = None
549
+ self.denoise_step = None
550
+ self.denoise_total_steps = None
551
+ self.condition = threading.Condition()
552
+
553
+ def add_event(self, event):
554
+ with self.condition:
555
+ # `at` is seconds since the job started (since it was created,
556
+ # for the events before that). Phases say what a step is waiting
557
+ # on; only a clock on each event says what it cost - the
558
+ # lead-in from `step_start` to the first `pipeline_step` on a
559
+ # reused pipeline is the number a "slow start" report needs
560
+ since = self.started_at if self.started_at is not None else self.created_at
561
+ self.events.append(
562
+ {"seq": len(self.events), "at": round(time.time() - since, 1), **event}
563
+ )
564
+ self._note_progress(event)
565
+ self.condition.notify_all()
566
+
567
+ def _note_progress(self, event):
568
+ """Fold one event into the running summary.
569
+
570
+ A single-step generation emits `generating` and then nothing until it
571
+ is done, so 'no new events' is the normal state of a healthy run and
572
+ says nothing about whether it is progressing. What answers that is
573
+ how long it has been that way, and how far into the denoise loop it
574
+ got - both of which are here rather than in the event log.
575
+ """
576
+ now = time.time()
577
+ self.last_event_at = now
578
+ kind = event.get("event")
579
+ if kind == "phase":
580
+ self.phase = event.get("phase")
581
+ self.phase_detail = event.get("detail")
582
+ self.phase_started_at = now
583
+ elif kind == "pipeline_step":
584
+ self.denoise_step = event.get("step")
585
+ self.denoise_total_steps = event.get("total_steps")
586
+ elif kind == "download_progress":
587
+ # Folded into phase_detail rather than a field of its own - a
588
+ # poller already reads phase_detail for what the loading phase
589
+ # is waiting on, and the next "phase" event (loading ending)
590
+ # overwrites it same as any other detail (#343)
591
+ self.phase_detail = format_progress(
592
+ event.get("repo_id"),
593
+ event.get("downloaded_bytes"),
594
+ event.get("bytes_per_second"),
595
+ event.get("seconds_since_bytes_changed"),
596
+ )
597
+ elif kind == "warning":
598
+ # Both channels, on purpose: the event log keeps the moment it
599
+ # happened, `warnings` keeps it where a caller who polled the
600
+ # finished job will actually look, since a warning about the
601
+ # artifact outlives the run that noticed it (#82). The step it
602
+ # fired in is the run's, not the warning's - the engine warns
603
+ # from inside a step without knowing which one it is.
604
+ #
605
+ # A phase-stall report (#176) is the exception: it is a moment,
606
+ # not a fact about the result - a 90 s cold load says "still in
607
+ # phase 'loading'" three times and then succeeds - so it stays
608
+ # in the event log only. `warnings` is the channel a consumer
609
+ # reads after the run, and the regression suites assert it is
610
+ # empty on a clean one.
611
+ message = event.get("message")
612
+ if message and event.get("kind") != "phase_stall":
613
+ named = f"{self.step_name}: {message}" if self.step_name else message
614
+ if named not in self.warnings:
615
+ self.warnings.append(named)
616
+ elif kind == "step_start":
617
+ self.step_name = event.get("step")
618
+ # A sub-workflow counts its own steps from zero; what a caller
619
+ # watching a composed run needs is where the run it queued has
620
+ # got to, so the parent's counter wins when the event carries
621
+ # one and the step name stays the child's (#90)
622
+ self.parent_step = event.get("parent_step")
623
+ self.step_index = event.get("parent_index", event.get("index"))
624
+ self.total_steps = event.get("parent_total_steps", event.get("total_steps"))
625
+ # A new step's denoise loop has not started; the previous step's
626
+ # count would read as this one's progress
627
+ self.denoise_step = None
628
+ self.denoise_total_steps = None
629
+
630
+ def progress(self):
631
+ """Where a running job has got to, or None for one that has not
632
+ started - a terminal job has a manifest, which is a better answer
633
+ than a stale phase, except for FAILED: the manifest is only the
634
+ steps that finished, not the one that was running when the job died,
635
+ and that phase (`loading` / `generating` / `decoding` / `saving`) is
636
+ the fastest way to tell what killed it without reading a traceback
637
+ (#269). Frozen at `finished_at` rather than read against the current
638
+ clock, so `seconds_in_phase` reports how long the dead step had been
639
+ running rather than growing forever after the job is long over."""
640
+ if self.last_event_at is None or self.status not in (RUNNING, FAILED):
641
+ return None
642
+ now = (
643
+ self.finished_at
644
+ if self.status == FAILED and self.finished_at
645
+ else time.time()
646
+ )
647
+ summary = {
648
+ "step": self.step_name,
649
+ # The step of the queued workflow the one above is running
650
+ # inside, for a composed run; null when they are the same thing
651
+ "parent_step": self.parent_step,
652
+ "step_index": self.step_index,
653
+ "total_steps": self.total_steps,
654
+ "phase": self.phase,
655
+ "phase_detail": self.phase_detail,
656
+ "seconds_in_phase": (
657
+ round(now - self.phase_started_at, 1) if self.phase_started_at else None
658
+ ),
659
+ # The one number that separates a slow run from a hung one -
660
+ # but only once the denoise loop is running, see below
661
+ "seconds_since_event": round(now - self.last_event_at, 1),
662
+ # Always present, null until the loop starts. A key that only
663
+ # appears once there is a count to report cannot be told apart
664
+ # from a key that is missing because nothing is happening: the
665
+ # lead-in to `generating` - encoding the prompt and any
666
+ # reference image or audio - is over a minute of silence on a
667
+ # large video model, and read as an absent counter it looks
668
+ # exactly like a wedged denoise loop. Null here means the loop
669
+ # has not started; a number that stops moving is the stuck one
670
+ "denoise_step": self.denoise_step,
671
+ "denoise_total_steps": self.denoise_total_steps,
672
+ }
673
+ return summary
674
+
675
+ def finish(self, status, error=None, traceback_text=None):
676
+ self.status = status
677
+ self.finished_at = time.time()
678
+ self.error = error
679
+ self.traceback = traceback_text
680
+ self.add_event({"event": "job_status", "status": status})
681
+
682
+ def events_after(self, after_seq):
683
+ # Clamped: an 'after' below -1 would slice from the END of the log
684
+ # (events[-4:] for after=-5) and silently drop the earlier events a
685
+ # client asking for everything expects
686
+ after_seq = max(after_seq, -1)
687
+ with self.condition:
688
+ return self.events[after_seq + 1 :]
689
+
690
+ def wait_for_event(self, after_seq, timeout):
691
+ """Block until an event past after_seq exists or the job ends."""
692
+ with self.condition:
693
+ if len(self.events) > after_seq + 1 or self.status in TERMINAL_STATES:
694
+ return
695
+ self.condition.wait(timeout)
696
+
697
+ def summary(self):
698
+ return {
699
+ "id": self.id,
700
+ "workflow": self.workflow_name,
701
+ "workflow_name": self.catalog_name,
702
+ "status": self.status,
703
+ "created_at": self.created_at,
704
+ "started_at": self.started_at,
705
+ "finished_at": self.finished_at,
706
+ # Which workspace this job runs in - a live job's spec may not
707
+ # carry one yet (e.g. a caller that never named a workspace),
708
+ # so it defaults the same way history's column does
709
+ "workspace": self.spec.get("workspace") or DEFAULT_WORKSPACE_NAME,
710
+ "run_id": self.run_id,
711
+ # The run's ordinal - 'v4' - so the job that just ran can be
712
+ # named the way the gallery will name it
713
+ "run_version": self.run_version,
714
+ "acknowledged": self.acknowledged,
715
+ }
716
+
717
+ def detail(self):
718
+ return {
719
+ **self.summary(),
720
+ "arguments": self.spec.get("arguments", {}),
721
+ "warnings": self.warnings,
722
+ "manifest": self.manifest,
723
+ "error": self.error,
724
+ "traceback": self.traceback,
725
+ "event_count": len(self.events),
726
+ "run_dir": self.run_dir,
727
+ "acknowledged_cost": self.spec.get("acknowledged_cost"),
728
+ "progress": self.progress(),
729
+ }
730
+
731
+
732
+ class JobManager:
733
+ """Serializes job execution onto the one GPU worker process."""
734
+
735
+ def __init__(
736
+ self,
737
+ output_dir,
738
+ log_level="INFO",
739
+ worker_manager=None,
740
+ history_path=None,
741
+ workflow_dir=None,
742
+ ):
743
+ self.output_dir = validate_output_path(output_dir, None)
744
+ # Confines workflow_path/base_dir/sub-workflow resolution for every
745
+ # job this manager submits - the server's configured workflow_dir
746
+ self.workflow_dir = workflow_dir
747
+ os.makedirs(self.output_dir, exist_ok=True)
748
+ self.log_level = log_level
749
+ self.worker_manager = worker_manager or WorkerManager()
750
+ self.history = JobHistory(history_path or resolve_path("jobs.sqlite"))
751
+ self.jobs = {}
752
+ self.last_memory = None
753
+ self.last_memory_at = None
754
+ # Reentrant: cancel() finishes a queued job while holding it, and
755
+ # _finish's terminal-job trim needs it again on the same thread
756
+ self._lock = threading.RLock() # guards job state transitions
757
+ # Pending job ids in run order - a list, not a Queue, so the queue
758
+ # can be reordered while jobs wait
759
+ self._pending = []
760
+ self._wake = threading.Condition(self._lock)
761
+ self._worker_lock = threading.Lock() # guards worker communication
762
+ self._current_job_id = None
763
+ self._stop = threading.Event()
764
+ self._runner = threading.Thread(
765
+ target=self._run_loop, daemon=True, name="job-runner"
766
+ )
767
+ self._runner.start()
768
+
769
+ # ------------------------------------------------------------- submission
770
+
771
+ def submit(
772
+ self,
773
+ workflow_path=None,
774
+ workflow=None,
775
+ arguments=None,
776
+ base_dir=None,
777
+ workflow_dir=None,
778
+ output_dir=None,
779
+ asset_dir=None,
780
+ workspace=None,
781
+ catalog_name=None,
782
+ acknowledged=ACK_NONE,
783
+ acknowledged_cost=None,
784
+ ):
785
+ """Validate a job request and queue it. Raises ValueError on a bad
786
+ request so the HTTP layer can answer 400 before anything runs.
787
+
788
+ `workflow_dir` overrides this job's confinement root for a workflow
789
+ that lives outside the writable directory - an example or a builtin,
790
+ which the caller has already resolved against the search path. The
791
+ worker re-validates against whatever this job records, so the
792
+ override travels with the job rather than widening the manager.
793
+
794
+ `output_dir`, `asset_dir` and `workspace` name which workspace this
795
+ job runs in. They travel with the job for the same reason: one
796
+ server holds several workspaces, and the process-wide roots would
797
+ make every job belong to whichever one was configured at startup.
798
+
799
+ `catalog_name` is the listing name the caller resolved `workflow_path`
800
+ from, kept for history; None for an inline definition.
801
+
802
+ `acknowledged` is the form of cost acknowledgement the caller gave
803
+ (none/boolean/bound) and `acknowledged_cost` the bound object - both
804
+ recorded, neither checked here; the route checks (#85).
805
+ """
806
+ arguments = arguments or {}
807
+ if (workflow_path is None) == (workflow is None):
808
+ raise ValueError("Provide exactly one of workflow_path or workflow")
809
+
810
+ confinement = workflow_dir or self.workflow_dir
811
+ job_output_dir = (
812
+ validate_output_path(output_dir, None) if output_dir else self.output_dir
813
+ )
814
+ os.makedirs(job_output_dir, exist_ok=True)
815
+
816
+ if workflow_path is not None:
817
+ # Loads and schema-validates now - a bad path or file fails the
818
+ # request, not the queue. Checked against the caller's arguments,
819
+ # not the document alone - a bare validate() checks the document
820
+ # with no arguments and so could refuse a run _candidate_for had
821
+ # already accepted for the same call (#415, the run_workflow
822
+ # mirror of #414)
823
+ loaded = workflow_from_file(workflow_path, job_output_dir, confinement)
824
+ errors = loaded.validation_errors(arguments=arguments)
825
+ if errors:
826
+ raise Exception(format_validation_errors(errors))
827
+ spec = {
828
+ "workflow_path": workflow_path,
829
+ "workflow_name": loaded.name,
830
+ "arguments": arguments,
831
+ "workflow_dir": confinement,
832
+ }
833
+ else:
834
+ # workflow_from_definition validates base_dir - it is HTTP-supplied
835
+ # path input and goes through the security layer like every path
836
+ loaded = workflow_from_definition(
837
+ copy.deepcopy(workflow), job_output_dir, base_dir, confinement
838
+ )
839
+ errors = loaded.validation_errors(arguments=arguments)
840
+ if errors:
841
+ raise Exception(format_validation_errors(errors))
842
+ spec = {
843
+ "workflow": workflow,
844
+ # Must match workflow_from_definition's fallback - the worker
845
+ # re-validates this against workflow_dir
846
+ "base_dir": base_dir
847
+ or (os.path.abspath(confinement) if confinement else os.getcwd()),
848
+ "workflow_name": loaded.name,
849
+ "arguments": arguments,
850
+ # Must be the same root the worker re-validates base_dir
851
+ # against (workflow_from_definition -> validate_path) - this
852
+ # job's own confinement, not the manager's process-wide
853
+ # default, or a named workspace's inline job fails after a
854
+ # 201 the moment base_dir and workflow_dir disagree
855
+ "workflow_dir": confinement,
856
+ }
857
+
858
+ # Which workspace this job runs in, and the roots that follow from
859
+ # it - recorded on the job so history, the worker command and a
860
+ # rerun all agree without re-deriving them
861
+ spec["workspace"] = workspace
862
+ spec["catalog_name"] = catalog_name
863
+ # The acknowledgement form travels with the job so history can say
864
+ # whether this run was consented to at its actual size (#85)
865
+ spec["acknowledged"] = acknowledged
866
+ if acknowledged_cost is not None:
867
+ spec["acknowledged_cost"] = acknowledged_cost
868
+ spec["output_dir"] = job_output_dir
869
+ if asset_dir:
870
+ spec["asset_dir"] = asset_dir
871
+
872
+ # The caller's own arguments, checked against the variables this
873
+ # workflow declares. set_variables makes the same check at the top of
874
+ # the run, so a bad name failed a job that had already been queued -
875
+ # and a workflow declaring no variables dropped every argument in
876
+ # silence. Refused here instead, while it is still a 400
877
+ problems = argument_errors(loaded.workflow_definition, arguments)
878
+ if problems:
879
+ raise ValueError(
880
+ "; ".join(
881
+ f"{problem['path']}: {problem['message']}" for problem in problems
882
+ )
883
+ )
884
+
885
+ # Signature-level check of pipeline arguments - the typo that would
886
+ # otherwise be a TypeError after the model loads becomes a warning
887
+ # the client sees at submission
888
+ spec["warnings"] = workflow_argument_warnings(
889
+ loaded.workflow_definition, arguments
890
+ )
891
+
892
+ job = Job(spec)
893
+ with self._lock:
894
+ self.jobs[job.id] = job
895
+ job.add_event({"event": "job_status", "status": QUEUED})
896
+ with self._wake:
897
+ self._pending.append(job.id)
898
+ self._wake.notify()
899
+ logger.info(f"Queued job {job.id} for workflow {job.workflow_name}")
900
+ return job
901
+
902
+ def get(self, job_id):
903
+ """A live Job, or a historical detail dict for a finished past run."""
904
+ job = self.jobs.get(job_id)
905
+ if job is not None:
906
+ return job
907
+ return self.history.get(job_id)
908
+
909
+ def definition(self, job_id):
910
+ """The workflow JSON a job ran, for a read-only view of it.
911
+
912
+ An inline definition comes straight from the spec; a job launched
913
+ from a path is re-read from disk, confined to the root the job ran
914
+ against. None when there is no such job, or when the file it named
915
+ has since moved, grown past the size limit or stopped parsing - a
916
+ graph of the run is a nicety, never a reason to fail the page.
917
+ """
918
+ job = self.jobs.get(job_id)
919
+ if job is not None:
920
+ spec = job.spec
921
+ else:
922
+ historical = self.history.get(job_id)
923
+ if historical is None:
924
+ return None
925
+ spec = historical.get("spec") or {}
926
+ inline = spec.get("workflow")
927
+ if inline is not None:
928
+ return copy.deepcopy(inline)
929
+ path = spec.get("workflow_path")
930
+ if not path:
931
+ return None
932
+ try:
933
+ validated = validate_workflow_path(
934
+ path, spec.get("workflow_dir") or self.workflow_dir
935
+ )
936
+ validate_json_size(validated)
937
+ with open(validated, "r") as file:
938
+ return json.load(file)
939
+ except (SecurityError, OSError, ValueError):
940
+ logger.debug(f"No workflow definition available for job {job_id}")
941
+ return None
942
+
943
+ def realized(self, job_id):
944
+ """The realized workflow a job ran, or None when the job predates
945
+ run tracking or its run directory no longer holds the file.
946
+
947
+ Read from the job's own output directory, not the manager's: one
948
+ server holds several workspaces, and a job carries the root it ran
949
+ against. The join is confined to that root, so a run_dir read back
950
+ out of the database cannot name anything outside it.
951
+ """
952
+ job = self.jobs.get(job_id)
953
+ if job is not None:
954
+ run_dir = job.run_dir
955
+ output_dir = job.spec.get("output_dir") or self.output_dir
956
+ else:
957
+ historical = self.history.get(job_id)
958
+ if historical is None:
959
+ return None
960
+ run_dir = historical.get("run_dir")
961
+ output_dir = (historical.get("spec") or {}).get(
962
+ "output_dir"
963
+ ) or self.output_dir
964
+ if not run_dir:
965
+ return None
966
+ try:
967
+ root = validate_output_path(output_dir, None)
968
+ path = validate_path(os.path.join(root, run_dir, REALIZED_FILE_NAME), root)
969
+ validate_json_size(path)
970
+ with open(path, "r") as file:
971
+ return json.load(file)
972
+ except (SecurityError, OSError, ValueError) as e:
973
+ logger.debug(f"No realized workflow for job {job_id}: {e}")
974
+ return None
975
+
976
+ def seed_variable(self, job_id):
977
+ """The variable this job's workflow draws its seed from, or None.
978
+
979
+ Read from the workflow as written, never from the realized copy the
980
+ run wrote: realization pins the top-level seed to the integer the run
981
+ used, so a realized workflow always looks like it names a literal.
982
+
983
+ None means a new-seed rerun has nowhere to put one - either the seed
984
+ is a literal (an argument cannot override it) or the workflow names
985
+ no seed at all, in which case every run already draws a fresh one and
986
+ the step cache is off.
987
+ """
988
+ definition = self.definition(job_id)
989
+ seed = (definition or {}).get("seed")
990
+ if not isinstance(seed, str) or not seed.startswith(VARIABLE_PREFIX):
991
+ return None
992
+ name = seed.removeprefix(VARIABLE_PREFIX)
993
+ return name if name in (definition.get("variables") or {}) else None
994
+
995
+ def rerun_spec(self, job_id):
996
+ """The spec and arguments a rerun of `job_id` would submit, as
997
+ (spec, arguments), or None for an unknown job - split from rerun()
998
+ so a route can plan the run before queuing it (#85)."""
999
+ job = self.jobs.get(job_id)
1000
+ if job is not None:
1001
+ spec = {key: job.spec[key] for key in RERUN_SPEC_KEYS if key in job.spec}
1002
+ return spec, job.spec.get("arguments", {})
1003
+ historical = self.history.get(job_id)
1004
+ if historical is None:
1005
+ return None
1006
+ spec = {
1007
+ key: historical["spec"][key]
1008
+ for key in RERUN_SPEC_KEYS
1009
+ if key in historical["spec"]
1010
+ }
1011
+ return spec, historical["arguments"]
1012
+
1013
+ def rerun(
1014
+ self, job_id, new_seed=False, acknowledged=ACK_NONE, acknowledged_cost=None
1015
+ ):
1016
+ """Queue a fresh job from a previous job's spec.
1017
+
1018
+ Every root the original ran against (workflow_dir/output_dir/
1019
+ asset_dir/workspace) rides along, not just the workflow identity -
1020
+ otherwise a rerun of a job from a named workspace would fall back to
1021
+ the manager's process-wide default and silently run somewhere else.
1022
+
1023
+ `new_seed` draws a fresh seed into the workflow's seed variable. A
1024
+ plain rerun of a seeded workflow repeats its arguments exactly, which
1025
+ makes every step a step-cache hit: it republishes the earlier run's
1026
+ files in a fraction of a second and generates nothing. That is the
1027
+ cache doing its job - the same seed and the same inputs would produce
1028
+ the same pixels - so the way to actually get another image is to
1029
+ change the seed, and this is that.
1030
+
1031
+ `acknowledged` and `acknowledged_cost` are this request's own; the
1032
+ original's bound object rides along in the spec for the record when
1033
+ the request brought none.
1034
+ """
1035
+ prepared = self.rerun_spec(job_id)
1036
+ if prepared is None:
1037
+ return None
1038
+ spec, arguments = prepared
1039
+
1040
+ if new_seed:
1041
+ variable = self.seed_variable(job_id)
1042
+ if variable is None:
1043
+ raise ValueError(
1044
+ "This workflow does not draw its seed from a variable, so "
1045
+ "a rerun cannot change it. A workflow with no seed at all "
1046
+ "already draws a fresh one every run."
1047
+ )
1048
+ # Bounded so the number survives its trip through a browser as
1049
+ # JSON - see SEED_BITS
1050
+ arguments = {**arguments, variable: secrets.randbits(SEED_BITS)}
1051
+
1052
+ workspace = spec.get("workspace")
1053
+ if (
1054
+ workspace
1055
+ and workspace != DEFAULT_WORKSPACE_NAME
1056
+ and spec.get("output_dir")
1057
+ and not os.path.isdir(spec["output_dir"])
1058
+ ):
1059
+ raise ValueError(f"Workspace '{workspace}' the job ran in no longer exists")
1060
+
1061
+ return self.submit(
1062
+ workflow_path=spec.get("workflow_path"),
1063
+ workflow=spec.get("workflow"),
1064
+ arguments=arguments,
1065
+ base_dir=spec.get("base_dir"),
1066
+ workflow_dir=spec.get("workflow_dir"),
1067
+ output_dir=spec.get("output_dir"),
1068
+ asset_dir=spec.get("asset_dir"),
1069
+ workspace=workspace,
1070
+ catalog_name=spec.get("catalog_name"),
1071
+ acknowledged=acknowledged,
1072
+ acknowledged_cost=(
1073
+ acknowledged_cost
1074
+ if acknowledged_cost is not None
1075
+ else spec.get("acknowledged_cost")
1076
+ ),
1077
+ )
1078
+
1079
+ def queue_position(self, job_id):
1080
+ """Index in the waiting queue, or None when the job is not queued."""
1081
+ with self._lock:
1082
+ return self._pending.index(job_id) if job_id in self._pending else None
1083
+
1084
+ def describe(self, job):
1085
+ """A live job's detail plus its queue position while it waits - what
1086
+ the per-job endpoints return, so a client holding one job can say
1087
+ where it stands without fetching the whole list."""
1088
+ detail = job.detail()
1089
+ position = self.queue_position(job.id)
1090
+ if position is not None:
1091
+ detail["queue_position"] = position
1092
+ return detail
1093
+
1094
+ def list(self, workspace=None, statuses=None):
1095
+ """All jobs, live and historical, sorted by creation. `workspace` filters to one workspace; omitted, the list
1096
+ spans every workspace the server holds, unchanged from before
1097
+ workspaces existed. `statuses` filters to a set of job states
1098
+ ('queued', 'running', 'succeeded', 'failed', 'cancelled'); omitted,
1099
+ every state is listed."""
1100
+ statuses = set(statuses) if statuses else None
1101
+ with self._lock:
1102
+ live = sorted(self.jobs.values(), key=lambda j: j.created_at)
1103
+ positions = {job_id: i for i, job_id in enumerate(self._pending)}
1104
+ live_ids = {job.id for job in live}
1105
+ summaries = []
1106
+ for job in live:
1107
+ summary = job.summary()
1108
+ if workspace and summary["workspace"] != workspace:
1109
+ continue
1110
+ if statuses and summary["status"] not in statuses:
1111
+ continue
1112
+ if job.id in positions:
1113
+ summary["queue_position"] = positions[job.id]
1114
+ summaries.append(summary)
1115
+ for historical in self.history.recent_summaries(
1116
+ workspace=workspace, statuses=statuses
1117
+ ):
1118
+ if historical["id"] not in live_ids:
1119
+ summaries.append(historical)
1120
+ summaries.sort(key=lambda summary: summary["created_at"] or 0)
1121
+ return summaries
1122
+
1123
+ # ------------------------------------------------------------ cancel/stop
1124
+
1125
+ def cancel(self, job_id):
1126
+ """Cancel a queued or running job. Returns the job's status after the
1127
+ request, or None for an unknown job."""
1128
+ job = self.jobs.get(job_id)
1129
+ if job is None:
1130
+ return None
1131
+ with self._lock:
1132
+ if job.status in TERMINAL_STATES:
1133
+ return job.status
1134
+ if job.status == QUEUED:
1135
+ if job.id in self._pending:
1136
+ self._pending.remove(job.id)
1137
+ self._finish(job, CANCELLED)
1138
+ return job.status
1139
+ if job.status == RUNNING and self._current_job_id == job.id:
1140
+ try:
1141
+ self.worker_manager.cancel()
1142
+ except Exception as e:
1143
+ logger.warning(f"Could not send cancel for job {job_id}: {e}")
1144
+ return job.status
1145
+
1146
+ def move(self, job_id, direction):
1147
+ """Reorder a queued job: 'up'/'down' swap with a neighbour,
1148
+ 'front'/'back' go to the ends. Returns the new pending order, or
1149
+ None for a job that is not queued (finished, running, unknown)."""
1150
+ if direction not in ("up", "down", "front", "back"):
1151
+ raise ValueError(f"Unknown queue direction '{direction}'")
1152
+ with self._lock:
1153
+ if job_id not in self._pending:
1154
+ return None
1155
+ index = self._pending.index(job_id)
1156
+ self._pending.pop(index)
1157
+ if direction == "front":
1158
+ index = 0
1159
+ elif direction == "back":
1160
+ index = len(self._pending)
1161
+ elif direction == "up":
1162
+ index = max(0, index - 1)
1163
+ else:
1164
+ index = min(len(self._pending), index + 1)
1165
+ self._pending.insert(index, job_id)
1166
+ return list(self._pending)
1167
+
1168
+ def shutdown(self):
1169
+ self._stop.set()
1170
+ with self._wake:
1171
+ self._wake.notify_all()
1172
+ self._runner.join(timeout=5)
1173
+ self.worker_manager.shutdown_worker()
1174
+
1175
+ # ---------------------------------------------------------------- runner
1176
+
1177
+ def _run_loop(self):
1178
+ while not self._stop.is_set():
1179
+ with self._wake:
1180
+ while not self._pending and not self._stop.is_set():
1181
+ self._wake.wait()
1182
+ if self._stop.is_set():
1183
+ return
1184
+ job_id = self._pending.pop(0)
1185
+ job = self.jobs.get(job_id)
1186
+ if job is None or job.status != QUEUED:
1187
+ continue # cancelled while waiting
1188
+ self._run_job(job)
1189
+
1190
+ def _finish(self, job, status, error=None, traceback_text=None):
1191
+ job.finish(status, error=error, traceback_text=traceback_text)
1192
+ try:
1193
+ self.history.record(job)
1194
+ except Exception as e:
1195
+ logger.warning(f"Could not persist job {job.id}: {e}")
1196
+ self._trim_terminal_jobs()
1197
+
1198
+ def _trim_terminal_jobs(self):
1199
+ """Drop the oldest finished jobs from memory - history has them, and
1200
+ get()/list() fall through to it. Recent ones stay for event replay."""
1201
+ with self._lock:
1202
+ terminal = [
1203
+ job
1204
+ for job in sorted(self.jobs.values(), key=lambda j: j.created_at)
1205
+ if job.status in TERMINAL_STATES
1206
+ ]
1207
+ for job in terminal[:-TERMINAL_JOBS_KEPT]:
1208
+ del self.jobs[job.id]
1209
+
1210
+ def _run_job(self, job):
1211
+ with self._worker_lock:
1212
+ with self._lock:
1213
+ if job.status != QUEUED:
1214
+ return
1215
+ job.status = RUNNING
1216
+ job.started_at = time.time()
1217
+ self._current_job_id = job.id
1218
+ job.add_event({"event": "job_status", "status": RUNNING})
1219
+ try:
1220
+ self.worker_manager.ensure_worker(self.log_level)
1221
+ command = {
1222
+ "type": "execute",
1223
+ "arguments": job.spec["arguments"],
1224
+ # The job's own roots, so a job queued for one workspace
1225
+ # still runs in it after the manager has served another
1226
+ "output_dir": job.spec.get("output_dir") or self.output_dir,
1227
+ "log_level": self.log_level,
1228
+ }
1229
+ if job.spec.get("asset_dir"):
1230
+ command["asset_dir"] = job.spec["asset_dir"]
1231
+ if "workflow_path" in job.spec:
1232
+ command["workflow_path"] = job.spec["workflow_path"]
1233
+ else:
1234
+ command["workflow"] = job.spec["workflow"]
1235
+ command["base_dir"] = job.spec["base_dir"]
1236
+ command["workflow_dir"] = job.spec.get("workflow_dir")
1237
+ self.worker_manager.send_command(command)
1238
+ outcome = self._consume_results(job)
1239
+ except Exception as e:
1240
+ logger.error(f"Job {job.id} failed: {e}", exc_info=True)
1241
+ outcome = (FAILED, str(e), None)
1242
+ finally:
1243
+ # Cleared BEFORE the terminal status becomes visible - a
1244
+ # client seeing "succeeded" must find the manager idle
1245
+ with self._lock:
1246
+ self._current_job_id = None
1247
+ if job.status not in TERMINAL_STATES:
1248
+ status, error, traceback_text = outcome
1249
+ self._finish(job, status, error=error, traceback_text=traceback_text)
1250
+
1251
+ def _record_manifest(self, job, message):
1252
+ """What the run wrote, named the way clients address outputs.
1253
+
1254
+ Recorded for a failed or cancelled run as well as a successful one -
1255
+ the files the steps before the stop wrote are on disk either way,
1256
+ and a manifest that omits them is the difference between "this run
1257
+ produced nothing" and "this run produced four of five shots"
1258
+ (T015)."""
1259
+ job.manifest = self._relative_manifest(
1260
+ message.get("manifest", []), job.spec.get("output_dir")
1261
+ )
1262
+
1263
+ def _relative_manifest(self, manifest, output_dir=None):
1264
+ """A manifest list with every entry's 'files' relativised - the
1265
+ rendering `get_job` and `step_end`/`workflow_end` events must all
1266
+ agree on (#284)."""
1267
+ return [
1268
+ (
1269
+ {
1270
+ **entry,
1271
+ "files": self._relative_output_names(entry["files"], output_dir),
1272
+ }
1273
+ if "files" in entry
1274
+ else entry
1275
+ )
1276
+ for entry in manifest
1277
+ ]
1278
+
1279
+ def _relative_output_names(self, paths, output_dir=None):
1280
+ """The worker reports absolute paths; clients build '/outputs/<name>'
1281
+ URLs, and a run writes under '<output_dir>/<identity>/<run id>/'
1282
+ (dw/workflow.py's effective_output_dir) - so every file is reported
1283
+ by its name relative to the output directory of the job that wrote
1284
+ it, with forward slashes. A path outside it (a task step writing
1285
+ elsewhere) is left as it came.
1286
+
1287
+ The job's own directory, not the manager's: a job in a named
1288
+ workspace writes under that workspace, and naming it relative to the
1289
+ default workspace would produce '../<name>/outputs/...' - a path, not
1290
+ a name."""
1291
+ root = output_dir or self.output_dir
1292
+ names = []
1293
+ for path in paths:
1294
+ relative = os.path.relpath(path, root)
1295
+ if relative.startswith(".."):
1296
+ names.append(path)
1297
+ else:
1298
+ names.append(relative.replace(os.sep, "/"))
1299
+ return names
1300
+
1301
+ def _consume_results(self, job):
1302
+ """Read worker messages until the run ends; returns the terminal
1303
+ (status, error, traceback) for _run_job to apply once the manager
1304
+ no longer counts the job as current."""
1305
+ while True:
1306
+ try:
1307
+ message = self.worker_manager.get_result()
1308
+ except RuntimeError as e:
1309
+ # The worker died without managing to send anything - a
1310
+ # signal, not an exception, so worker_main's handler never
1311
+ # ran and there is no traceback to be had. The exit code is
1312
+ # the only diagnosis available, and marking the crash here
1313
+ # matters beyond this job: the manager would otherwise go on
1314
+ # believing a dead process is active, and every later call
1315
+ # that talks to it (memory_status above all) would fail
1316
+ # against a queue nobody is reading
1317
+ detail = self.worker_manager.crash_details()
1318
+ self.worker_manager.mark_crashed()
1319
+ reason = f"Worker process died: {detail or e}"
1320
+ logger.error(f"Job {job.id}: {reason}")
1321
+ return (FAILED, reason, None)
1322
+ message_type = message.get("type")
1323
+
1324
+ if message_type == "progress":
1325
+ event = {k: v for k, v in message.items() if k != "type"}
1326
+ if "files" in event:
1327
+ event["files"] = self._relative_output_names(
1328
+ event["files"], job.spec.get("output_dir")
1329
+ )
1330
+ if "manifest" in event:
1331
+ # workflow_end carries the run's full manifest nested
1332
+ # under this key - it must match get_job's rendering of
1333
+ # the same list rather than leaking absolute paths (#284)
1334
+ event["manifest"] = self._relative_manifest(
1335
+ event["manifest"], job.spec.get("output_dir")
1336
+ )
1337
+ if event.get("event") == "run_start":
1338
+ job.run_id = event.get("run_id")
1339
+ job.run_dir = event.get("run_dir")
1340
+ job.run_version = event.get("version")
1341
+ job.add_event(event)
1342
+ elif message_type in ("output", "workflow_loaded"):
1343
+ text = message.get("message") or message.get("workflow_name", "")
1344
+ job.add_event({"event": "log", "message": text})
1345
+ elif message_type == "memory_info":
1346
+ # One per phase boundary now, not just once post-run (#273) -
1347
+ # each folds into the cached reading memory_status() answers
1348
+ # from while the job is busy, which is what makes that call
1349
+ # fresh instead of a refusal for the run's whole duration
1350
+ self._record_memory(message.get("info"))
1351
+ job.add_event({"event": "memory", "info": self.last_memory})
1352
+ # The worker's own high-water mark, latest reading wins (it
1353
+ # is monotonic for the process' life) - persisted as a real
1354
+ # column rather than only inside the trimmed event tail (#243)
1355
+ info = message.get("info") or {}
1356
+ peak = info.get("host_memory_peak_rss_mb")
1357
+ if peak is not None:
1358
+ job.host_memory_peak_rss_mb = peak
1359
+ # This job's own contribution to that process-lifetime peak,
1360
+ # computed against its own baseline (#272) - max() is
1361
+ # defensive; by construction each reading only grows
1362
+ job_peak = info.get("host_memory_job_peak_rss_mb")
1363
+ if job_peak is not None:
1364
+ job.host_memory_job_peak_rss_mb = max(
1365
+ job_peak, job.host_memory_job_peak_rss_mb or 0
1366
+ )
1367
+ elif message_type == "success":
1368
+ self._record_manifest(job, message)
1369
+ return (SUCCEEDED, None, None)
1370
+ elif message_type == "cancelled":
1371
+ self._record_manifest(job, message)
1372
+ return (CANCELLED, None, None)
1373
+ elif message_type == "error":
1374
+ # A failed run's steps too: the ones before the failure wrote
1375
+ # real files, and a job that reports an empty manifest hides
1376
+ # them behind the error that stopped the run
1377
+ self._record_manifest(job, message)
1378
+ return (
1379
+ FAILED,
1380
+ message.get("message"),
1381
+ message.get("traceback"),
1382
+ )
1383
+ elif message_type == "worker_crashed":
1384
+ self.worker_manager.mark_crashed()
1385
+ return (
1386
+ FAILED,
1387
+ f"Worker crashed: {message.get('message')}",
1388
+ message.get("traceback"),
1389
+ )
1390
+ else:
1391
+ logger.warning(f"Unknown worker message type: {message_type}")
1392
+
1393
+ def is_busy(self):
1394
+ """True while a job is running or queued - the window in which the
1395
+ worker may be reading model files a cache delete would rip out."""
1396
+ with self._lock:
1397
+ if self._current_job_id is not None:
1398
+ return True
1399
+ return any(job.status == QUEUED for job in self.jobs.values())
1400
+
1401
+ def restart_worker_if_idle(self):
1402
+ """Shut the idle worker down so its next start picks up upgraded
1403
+ imports; the next job respawns it via ensure_worker. Returns False
1404
+ without touching a busy worker - a run in flight keeps the version
1405
+ it started with."""
1406
+ if self.is_busy():
1407
+ return False
1408
+ if not self._worker_lock.acquire(timeout=2):
1409
+ return False
1410
+ try:
1411
+ self.worker_manager.shutdown_worker()
1412
+ return True
1413
+ finally:
1414
+ self._worker_lock.release()
1415
+
1416
+ # ---------------------------------------------------------------- memory
1417
+
1418
+ def _record_memory(self, info):
1419
+ """Remember a reading and when it was taken, so a later cached answer
1420
+ can say how old it is."""
1421
+ self.last_memory = info
1422
+ self.last_memory_at = time.time() if info is not None else None
1423
+
1424
+ def _cached_memory(self, reason):
1425
+ """The last reading, labelled with why it is not a live one. A caller
1426
+ comparing two readings must compare only `live: true` ones - a cached
1427
+ `info` was taken at another moment, and while a job loads a model it
1428
+ understates what is resident by however much has loaded since."""
1429
+ info = self.last_memory
1430
+ age = None
1431
+ if info is not None and self.last_memory_at is not None:
1432
+ age = round(time.time() - self.last_memory_at, 1)
1433
+ return {
1434
+ "live": False,
1435
+ "info": info,
1436
+ "stale": info is not None,
1437
+ "reason": reason,
1438
+ "age_seconds": age,
1439
+ }
1440
+
1441
+ def probe_cache(self, command, timeout=5):
1442
+ """Which steps the worker's step cache would serve for `command` (the
1443
+ fields an execute command carries, minus its type), or None when the
1444
+ answer cannot be had right now - a job is running, the worker is
1445
+ busy, or it did not answer in time. Never blocks a request behind a
1446
+ running job, for the same reason memory_status does not.
1447
+
1448
+ No worker running is a definite answer, not an unknown one: the
1449
+ cache lives in the worker process, so a worker that is not running
1450
+ holds nothing.
1451
+ """
1452
+ if self._current_job_id is not None:
1453
+ return None
1454
+ if not self.worker_manager.worker_active:
1455
+ return []
1456
+ if not self._worker_lock.acquire(timeout=2):
1457
+ return None
1458
+ # A probe that timed out still answers eventually, onto the same
1459
+ # queue the next probe reads - so each carries an id and a reader
1460
+ # discards every reply that is not its own, rather than reporting
1461
+ # the previous workflow's hit list as this plan's
1462
+ probe_id = uuid.uuid4().hex
1463
+ deadline = time.monotonic() + timeout
1464
+ try:
1465
+ self.worker_manager.send_command(
1466
+ {"type": "probe_cache", "probe_id": probe_id, **command}
1467
+ )
1468
+ while True:
1469
+ remaining = deadline - time.monotonic()
1470
+ if remaining <= 0:
1471
+ return None
1472
+ result = self.worker_manager.get_result(timeout=remaining)
1473
+ if (
1474
+ result.get("type") == "probe_cache"
1475
+ and result.get("probe_id") == probe_id
1476
+ ):
1477
+ break
1478
+ logger.debug(f"Discarding a stale worker message: {result.get('type')}")
1479
+ except (RuntimeError, queue.Empty) as e:
1480
+ logger.debug(f"Worker did not answer the cache probe: {e}")
1481
+ return None
1482
+ finally:
1483
+ self._worker_lock.release()
1484
+ cached = result.get("cached")
1485
+ return list(cached) if isinstance(cached, list) else None
1486
+
1487
+ def memory_status(self, timeout=5):
1488
+ """Live memory stats when the worker is idle; the run's last report
1489
+ while it is busy. The lock acquire is bounded: the runner holds
1490
+ _worker_lock for a job's whole duration, and a poll that raced a job
1491
+ start must fall back to the cached reading, not block for hours.
1492
+
1493
+ `live` says whether `info` was measured by this call. `stale` and
1494
+ `reason` say why it was not, and `age_seconds` how old the cached
1495
+ reading is; `info` is null when there has never been a reading, which
1496
+ means nothing is resident rather than that the answer is unknown."""
1497
+ if self._current_job_id is not None:
1498
+ return self._cached_memory("job_running")
1499
+ if not self.worker_manager.worker_active:
1500
+ return self._cached_memory("worker_stopped")
1501
+ if not self._worker_lock.acquire(timeout=2):
1502
+ return self._cached_memory("worker_busy")
1503
+ try:
1504
+ self.worker_manager.send_command({"type": "memory_status"})
1505
+ result = self.worker_manager.get_result(timeout=timeout)
1506
+ except (RuntimeError, queue.Empty) as e:
1507
+ # A dead worker is exactly when the last reading taken before it
1508
+ # died is worth the most, so report that rather than failing the
1509
+ # request. Without this the caller gets a 503 at the one moment
1510
+ # it most wants a number
1511
+ detail = self.worker_manager.crash_details()
1512
+ logger.warning(f"Worker unavailable for memory status: {detail or e}")
1513
+ if detail is not None:
1514
+ # crash_details only answers for a process the OS has reaped,
1515
+ # so a worker that is merely slow to reply keeps its state -
1516
+ # a timeout is not evidence of death
1517
+ self.worker_manager.mark_crashed()
1518
+ return self._cached_memory("worker_unreachable")
1519
+ finally:
1520
+ self._worker_lock.release()
1521
+ if result.get("type") == "memory_status":
1522
+ self._record_memory(result.get("info"))
1523
+ return {
1524
+ "live": True,
1525
+ "info": self.last_memory,
1526
+ "stale": False,
1527
+ "reason": None,
1528
+ "age_seconds": 0.0,
1529
+ }
1530
+ return self._cached_memory("worker_unreachable")
1531
+
1532
+ def clear_memory(self, timeout=30):
1533
+ """Drop every loaded pipeline and the step cache, then report the
1534
+ memory reading taken right after. Callers must check `is_busy()`
1535
+ first - this does not itself refuse a running/queued job, and racing
1536
+ one would clear state a queued run still expects resident. The
1537
+ 30s timeout (vs. `memory_status`'s 5s) matches the REPL's `memory
1538
+ clear` (`repl_commands.py`): actually freeing CUDA memory takes
1539
+ longer than reading a counter does.
1540
+
1541
+ Returns the reading taken after the clear, or None when there was no
1542
+ worker to clear - nothing was resident in that case."""
1543
+ if not self.worker_manager.worker_active:
1544
+ # Nothing to clear, and not a fault: the pipelines and the step
1545
+ # cache both live in the worker process, so no worker running
1546
+ # means both are already gone. An on-demand worker is legitimately
1547
+ # absent on an idle server (#206), which this used to answer with
1548
+ # a 503 saying the worker was unavailable. None is "no reading
1549
+ # was taken", not a failure
1550
+ return None
1551
+ if not self._worker_lock.acquire(timeout=2):
1552
+ raise RuntimeError("worker busy")
1553
+ try:
1554
+ self.worker_manager.send_command({"type": "clear_memory"})
1555
+ result = self.worker_manager.get_result(timeout=timeout)
1556
+ finally:
1557
+ self._worker_lock.release()
1558
+ if result.get("type") != "memory_cleared":
1559
+ raise RuntimeError(f"unexpected worker reply: {result.get('type')}")
1560
+ self._record_memory(result.get("info"))
1561
+ return self.last_memory