diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/server/jobs.py
ADDED
|
@@ -0,0 +1,1561 @@
|
|
|
1
|
+
"""Job queue over the persistent worker process.
|
|
2
|
+
|
|
3
|
+
One runner thread executes jobs FIFO against the single GPU worker - the
|
|
4
|
+
same WorkerManager the REPL uses. Jobs collect their progress events with
|
|
5
|
+
sequence numbers so an SSE client can attach late (or reconnect) and replay
|
|
6
|
+
from where it left off.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
import copy
|
|
11
|
+
import json
|
|
12
|
+
import queue
|
|
13
|
+
import secrets
|
|
14
|
+
import sqlite3
|
|
15
|
+
import time
|
|
16
|
+
import uuid
|
|
17
|
+
import logging
|
|
18
|
+
import threading
|
|
19
|
+
|
|
20
|
+
from ..download_watch import format_progress
|
|
21
|
+
from ..repl_worker import WorkerManager
|
|
22
|
+
from ..workflow import SEED_BITS, workflow_from_file, workflow_from_definition
|
|
23
|
+
from ..introspection import workflow_argument_warnings
|
|
24
|
+
from ..schema import format_validation_errors
|
|
25
|
+
from ..variables import argument_errors
|
|
26
|
+
from ..security import (
|
|
27
|
+
SecurityError,
|
|
28
|
+
validate_json_size,
|
|
29
|
+
validate_output_path,
|
|
30
|
+
validate_path,
|
|
31
|
+
validate_workflow_path,
|
|
32
|
+
)
|
|
33
|
+
from ..realize import VARIABLE_PREFIX
|
|
34
|
+
from ..runs import REALIZED_FILE_NAME
|
|
35
|
+
from ..settings import resolve_path
|
|
36
|
+
from ..workspace import DEFAULT_WORKSPACE_NAME
|
|
37
|
+
from .observed_cost import EVENT_CAP, LOADING_MARKER
|
|
38
|
+
|
|
39
|
+
logger = logging.getLogger("dw")
|
|
40
|
+
|
|
41
|
+
QUEUED = "queued"
|
|
42
|
+
RUNNING = "running"
|
|
43
|
+
SUCCEEDED = "succeeded"
|
|
44
|
+
FAILED = "failed"
|
|
45
|
+
CANCELLED = "cancelled"
|
|
46
|
+
TERMINAL_STATES = (SUCCEEDED, FAILED, CANCELLED)
|
|
47
|
+
|
|
48
|
+
# Which form of cost acknowledgement a job was queued with (#85): none (the
|
|
49
|
+
# web UI and every HTTP caller that sends nothing), a bare boolean, or one
|
|
50
|
+
# bound to the plan that was validated
|
|
51
|
+
ACK_NONE = "none"
|
|
52
|
+
ACK_BOOLEAN = "boolean"
|
|
53
|
+
ACK_BOUND = "bound"
|
|
54
|
+
|
|
55
|
+
# The spec fields a rerun needs - shared by persistence and live rerun
|
|
56
|
+
RERUN_SPEC_KEYS = (
|
|
57
|
+
"workflow_path",
|
|
58
|
+
"workflow",
|
|
59
|
+
"base_dir",
|
|
60
|
+
# A rerun belongs in the workspace the original ran in, so the roots
|
|
61
|
+
# that decided that are part of what history keeps
|
|
62
|
+
"workspace",
|
|
63
|
+
"output_dir",
|
|
64
|
+
"asset_dir",
|
|
65
|
+
# so a rerun is attributed to the same catalog entry
|
|
66
|
+
"catalog_name",
|
|
67
|
+
"workflow_dir",
|
|
68
|
+
# what the original run was consented to, kept for the record - a
|
|
69
|
+
# rerun's own request decides its form
|
|
70
|
+
"acknowledged_cost",
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
# Finished jobs kept in memory for SSE replay grace; older ones live in
|
|
74
|
+
# history only, so a long-running server's memory stays bounded
|
|
75
|
+
TERMINAL_JOBS_KEPT = 20
|
|
76
|
+
|
|
77
|
+
# A long run emits thousands of progress events; the tail is what explains
|
|
78
|
+
# the outcome. Bounded so history stays a summary store, not an event log
|
|
79
|
+
MAX_PERSISTED_EVENTS = 200
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class JobHistory:
|
|
83
|
+
"""Finished jobs, persisted so the Jobs view survives server restarts.
|
|
84
|
+
|
|
85
|
+
Records land at terminal state only - a crash mid-run loses that run's
|
|
86
|
+
row, which is the right trade for never blocking the runner on disk.
|
|
87
|
+
The last MAX_PERSISTED_EVENTS progress events ride along, so a job can
|
|
88
|
+
still explain itself after a restart; everything earlier is dropped.
|
|
89
|
+
"""
|
|
90
|
+
|
|
91
|
+
def __init__(self, db_path):
|
|
92
|
+
self.db_path = str(db_path)
|
|
93
|
+
self._lock = threading.Lock()
|
|
94
|
+
with self._connect() as connection:
|
|
95
|
+
connection.execute("""CREATE TABLE IF NOT EXISTS jobs (
|
|
96
|
+
id TEXT PRIMARY KEY,
|
|
97
|
+
workflow TEXT,
|
|
98
|
+
status TEXT,
|
|
99
|
+
created_at REAL,
|
|
100
|
+
started_at REAL,
|
|
101
|
+
finished_at REAL,
|
|
102
|
+
arguments TEXT,
|
|
103
|
+
spec TEXT,
|
|
104
|
+
manifest TEXT,
|
|
105
|
+
warnings TEXT,
|
|
106
|
+
error TEXT,
|
|
107
|
+
events TEXT
|
|
108
|
+
)""")
|
|
109
|
+
# Databases written before events were persisted are missing the
|
|
110
|
+
# column; ALTER is the whole migration, and rows keep NULL
|
|
111
|
+
columns = {row[1] for row in connection.execute("PRAGMA table_info(jobs)")}
|
|
112
|
+
if "events" not in columns:
|
|
113
|
+
connection.execute("ALTER TABLE jobs ADD COLUMN events TEXT")
|
|
114
|
+
# Every row predating workspaces belongs to the default one -
|
|
115
|
+
# history that cannot say which workspace a job ran in stops
|
|
116
|
+
# making sense the moment there are two
|
|
117
|
+
if "workspace" not in columns:
|
|
118
|
+
connection.execute(
|
|
119
|
+
"ALTER TABLE jobs ADD COLUMN workspace TEXT DEFAULT 'default'"
|
|
120
|
+
)
|
|
121
|
+
connection.execute(
|
|
122
|
+
"UPDATE jobs SET workspace = 'default' WHERE workspace IS NULL"
|
|
123
|
+
)
|
|
124
|
+
# The catalog name the job was run from, beside `workflow` (the
|
|
125
|
+
# definition's id). Ids are not unique across a catalog forever;
|
|
126
|
+
# names are, and a later runtime-by-workflow join wants the exact
|
|
127
|
+
# one. Rows before this column stay NULL: old history is
|
|
128
|
+
# unjoinable, new history is exact
|
|
129
|
+
if "workflow_name" not in columns:
|
|
130
|
+
connection.execute("ALTER TABLE jobs ADD COLUMN workflow_name TEXT")
|
|
131
|
+
# Which run of the workflow this job was - the directory under the
|
|
132
|
+
# output root that holds its manifest and its realized workflow.
|
|
133
|
+
# NULL for every row predating run tracking, and the manager
|
|
134
|
+
# refuses to guess one from file paths
|
|
135
|
+
if "run_id" not in columns:
|
|
136
|
+
connection.execute("ALTER TABLE jobs ADD COLUMN run_id TEXT")
|
|
137
|
+
if "run_dir" not in columns:
|
|
138
|
+
connection.execute("ALTER TABLE jobs ADD COLUMN run_dir TEXT")
|
|
139
|
+
# That run's ordinal among the workflow's runs - the 'v4' the
|
|
140
|
+
# gallery shows. NULL before the column, and for a job that
|
|
141
|
+
# never opened a run
|
|
142
|
+
if "run_version" not in columns:
|
|
143
|
+
connection.execute("ALTER TABLE jobs ADD COLUMN run_version INTEGER")
|
|
144
|
+
# Which form of cost acknowledgement queued the job. Rows before
|
|
145
|
+
# the column are 'none' - nothing recorded is nothing recorded
|
|
146
|
+
if "acknowledged" not in columns:
|
|
147
|
+
connection.execute(
|
|
148
|
+
"ALTER TABLE jobs ADD COLUMN acknowledged TEXT DEFAULT 'none'"
|
|
149
|
+
)
|
|
150
|
+
# The worker's own high-water mark for this run (#243) - NULL for
|
|
151
|
+
# a row predating the column and for any run that never reported
|
|
152
|
+
# one (cancelled/errored before the worker's final memory_info)
|
|
153
|
+
if "host_memory_peak_rss_mb" not in columns:
|
|
154
|
+
connection.execute(
|
|
155
|
+
"ALTER TABLE jobs ADD COLUMN host_memory_peak_rss_mb REAL"
|
|
156
|
+
)
|
|
157
|
+
# This job's own contribution to that process-lifetime figure -
|
|
158
|
+
# growth since the job's first phase-boundary reading, or its
|
|
159
|
+
# current rss when it caused no growth (#272). NULL for a row
|
|
160
|
+
# predating the column and for any run that never got a
|
|
161
|
+
# memory_info message at all
|
|
162
|
+
if "host_memory_job_peak_rss_mb" not in columns:
|
|
163
|
+
connection.execute(
|
|
164
|
+
"ALTER TABLE jobs ADD COLUMN host_memory_job_peak_rss_mb REAL"
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
def _connect(self):
|
|
168
|
+
# WAL mode lets a reader (the web UI polling job status, an MCP
|
|
169
|
+
# get_job call) proceed without blocking behind whatever write the
|
|
170
|
+
# worker is mid-transaction on, and vice versa - the default
|
|
171
|
+
# rollback-journal mode takes a database-wide lock for the
|
|
172
|
+
# duration of a write. journal_mode is a property of the database
|
|
173
|
+
# file, not the connection, but PRAGMA is cheap and idempotent, so
|
|
174
|
+
# it is set on every connect rather than assumed to have stuck.
|
|
175
|
+
connection = sqlite3.connect(self.db_path, timeout=5)
|
|
176
|
+
connection.execute("PRAGMA journal_mode=WAL")
|
|
177
|
+
return connection
|
|
178
|
+
|
|
179
|
+
def record(self, job):
|
|
180
|
+
# The spec's workflow_name/warnings are derived; keep what rerun needs
|
|
181
|
+
rerun_spec = {key: job.spec[key] for key in RERUN_SPEC_KEYS if key in job.spec}
|
|
182
|
+
with self._lock, self._connect() as connection:
|
|
183
|
+
connection.execute(
|
|
184
|
+
"INSERT OR REPLACE INTO jobs (id, workflow, status, created_at,"
|
|
185
|
+
" started_at, finished_at, arguments, spec, manifest, warnings,"
|
|
186
|
+
" error, events, workspace, workflow_name, run_id, run_dir,"
|
|
187
|
+
" acknowledged, host_memory_peak_rss_mb,"
|
|
188
|
+
" host_memory_job_peak_rss_mb, run_version) VALUES"
|
|
189
|
+
" (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
190
|
+
(
|
|
191
|
+
job.id,
|
|
192
|
+
job.workflow_name,
|
|
193
|
+
job.status,
|
|
194
|
+
job.created_at,
|
|
195
|
+
job.started_at,
|
|
196
|
+
job.finished_at,
|
|
197
|
+
json.dumps(job.spec.get("arguments", {}), default=str),
|
|
198
|
+
json.dumps(rerun_spec, default=str),
|
|
199
|
+
json.dumps(job.manifest, default=str),
|
|
200
|
+
json.dumps(job.warnings, default=str),
|
|
201
|
+
job.error,
|
|
202
|
+
json.dumps(job.events[-MAX_PERSISTED_EVENTS:], default=str),
|
|
203
|
+
job.spec.get("workspace") or DEFAULT_WORKSPACE_NAME,
|
|
204
|
+
job.catalog_name,
|
|
205
|
+
job.run_id,
|
|
206
|
+
job.run_dir,
|
|
207
|
+
job.acknowledged,
|
|
208
|
+
# A test double or an older in-memory Job predating this
|
|
209
|
+
# column reports None here rather than failing record()
|
|
210
|
+
# (#243) - the same "absent means unknown" the column
|
|
211
|
+
# itself allows
|
|
212
|
+
getattr(job, "host_memory_peak_rss_mb", None),
|
|
213
|
+
getattr(job, "host_memory_job_peak_rss_mb", None),
|
|
214
|
+
getattr(job, "run_version", None),
|
|
215
|
+
),
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
def recent_summaries(self, limit=200, workspace=None, statuses=None):
|
|
219
|
+
"""Summary rows only - the jobs list is polled, and parsing four JSON
|
|
220
|
+
blobs per row just to show six scalars was pure waste.
|
|
221
|
+
|
|
222
|
+
`workspace` filters to one workspace's rows; omitted, history spans
|
|
223
|
+
all of them the way the list already did before workspaces existed.
|
|
224
|
+
`statuses` filters to a set of terminal states - in SQL rather than
|
|
225
|
+
over the returned rows, or the newest-first cap above would be
|
|
226
|
+
spent on rows the filter then drops.
|
|
227
|
+
"""
|
|
228
|
+
query = (
|
|
229
|
+
"SELECT id, workflow, status, created_at, started_at, finished_at,"
|
|
230
|
+
" workspace, workflow_name, run_id, acknowledged, run_version"
|
|
231
|
+
" FROM jobs"
|
|
232
|
+
)
|
|
233
|
+
params = []
|
|
234
|
+
clauses = []
|
|
235
|
+
if workspace:
|
|
236
|
+
clauses.append("workspace = ?")
|
|
237
|
+
params.append(workspace)
|
|
238
|
+
if statuses:
|
|
239
|
+
statuses = list(statuses)
|
|
240
|
+
placeholders = ", ".join("?" for _ in statuses)
|
|
241
|
+
clauses.append(f"status IN ({placeholders})")
|
|
242
|
+
params.extend(statuses)
|
|
243
|
+
if clauses:
|
|
244
|
+
query += " WHERE " + " AND ".join(clauses)
|
|
245
|
+
query += " ORDER BY created_at DESC LIMIT ?"
|
|
246
|
+
params.append(limit)
|
|
247
|
+
with self._lock, self._connect() as connection:
|
|
248
|
+
rows = connection.execute(query, params).fetchall()
|
|
249
|
+
return [
|
|
250
|
+
{
|
|
251
|
+
"id": row[0],
|
|
252
|
+
"workflow": row[1],
|
|
253
|
+
"status": row[2],
|
|
254
|
+
"created_at": row[3],
|
|
255
|
+
"started_at": row[4],
|
|
256
|
+
"finished_at": row[5],
|
|
257
|
+
"workspace": row[6] or DEFAULT_WORKSPACE_NAME,
|
|
258
|
+
"workflow_name": row[7],
|
|
259
|
+
"run_id": row[8],
|
|
260
|
+
"acknowledged": row[9] or ACK_NONE,
|
|
261
|
+
"run_version": row[10],
|
|
262
|
+
"historical": True,
|
|
263
|
+
}
|
|
264
|
+
for row in rows
|
|
265
|
+
]
|
|
266
|
+
|
|
267
|
+
def get(self, job_id):
|
|
268
|
+
with self._lock, self._connect() as connection:
|
|
269
|
+
row = connection.execute(
|
|
270
|
+
"SELECT id, workflow, status, created_at, started_at, finished_at,"
|
|
271
|
+
" arguments, spec, manifest, warnings, error, workspace,"
|
|
272
|
+
" workflow_name, run_id, run_dir, acknowledged, events,"
|
|
273
|
+
" run_version FROM jobs WHERE id = ?",
|
|
274
|
+
(job_id,),
|
|
275
|
+
).fetchone()
|
|
276
|
+
return self._to_detail(row) if row else None
|
|
277
|
+
|
|
278
|
+
def watermark(self):
|
|
279
|
+
"""How far the table has got - what a derived figure caches against.
|
|
280
|
+
|
|
281
|
+
A job landing changes every observed cost and changes no file, so an
|
|
282
|
+
mtime cache cannot see it (dw/server/observed_cost.py). Counted over
|
|
283
|
+
`workflow_name IS NOT NULL` rather than every row, because
|
|
284
|
+
`orphan_workflow_history` (#274) detaches a deleted workflow's rows by
|
|
285
|
+
clearing that column rather than deleting the row - an ordinary
|
|
286
|
+
`COUNT(*)` would not move, and `ObservedCosts` would keep serving the
|
|
287
|
+
purged figure until an unrelated job happened to land. Counting only
|
|
288
|
+
the joinable rows falls by exactly the amount a purge detaches, the
|
|
289
|
+
same as a prune lowering it.
|
|
290
|
+
"""
|
|
291
|
+
with self._lock, self._connect() as connection:
|
|
292
|
+
row = connection.execute(
|
|
293
|
+
"SELECT COUNT(*), MAX(finished_at) FROM jobs"
|
|
294
|
+
" WHERE workflow_name IS NOT NULL"
|
|
295
|
+
).fetchone()
|
|
296
|
+
return (row[0], row[1]) if row else (0, None)
|
|
297
|
+
|
|
298
|
+
def finished_runs(self):
|
|
299
|
+
"""Every successful, named run grouped by (workspace, workflow name),
|
|
300
|
+
as the rows an observed cost is derived from.
|
|
301
|
+
|
|
302
|
+
One query for the whole catalog rather than one per workflow. The
|
|
303
|
+
cold/warm split is decided in SQL on the persisted event tail - a
|
|
304
|
+
`loading` phase as `json.dumps` wrote it - so 200 events per row are
|
|
305
|
+
never parsed to answer a yes/no question, and whether that tail hit
|
|
306
|
+
its cap comes back too, because a run whose `loading` phase was
|
|
307
|
+
trimmed away has to count as neither rather than as warm.
|
|
308
|
+
|
|
309
|
+
Rows with no `workflow_name` (recorded before the column existed, or
|
|
310
|
+
run from an inline definition, or orphaned by `orphan_workflow_history`)
|
|
311
|
+
are unjoinable and left out. The workspace dimension is always in the
|
|
312
|
+
key here; whether a caller treats two workspaces as one history (a
|
|
313
|
+
shared catalog source, #154) or as separate (a workspace's own
|
|
314
|
+
writable copy, #274) is decided in `ObservedCosts.rows_for`, which is
|
|
315
|
+
the layer that knows which kind of source it was asked about.
|
|
316
|
+
"""
|
|
317
|
+
with self._lock, self._connect() as connection:
|
|
318
|
+
rows = connection.execute(
|
|
319
|
+
"SELECT workflow_name, workspace, started_at, finished_at,"
|
|
320
|
+
" arguments, manifest, INSTR(COALESCE(events, ''), ?) > 0,"
|
|
321
|
+
" COALESCE(json_array_length(COALESCE(events, '[]')), 0) >= ?,"
|
|
322
|
+
" host_memory_peak_rss_mb, host_memory_job_peak_rss_mb"
|
|
323
|
+
" FROM jobs WHERE status = ? AND workflow_name IS NOT NULL"
|
|
324
|
+
" AND started_at IS NOT NULL AND finished_at IS NOT NULL",
|
|
325
|
+
(LOADING_MARKER, EVENT_CAP, SUCCEEDED),
|
|
326
|
+
).fetchall()
|
|
327
|
+
grouped = {}
|
|
328
|
+
for (
|
|
329
|
+
name,
|
|
330
|
+
workspace,
|
|
331
|
+
started,
|
|
332
|
+
finished,
|
|
333
|
+
arguments,
|
|
334
|
+
manifest,
|
|
335
|
+
had_load,
|
|
336
|
+
at_cap,
|
|
337
|
+
peak_rss_mb,
|
|
338
|
+
job_peak_rss_mb,
|
|
339
|
+
) in rows:
|
|
340
|
+
key = (workspace or DEFAULT_WORKSPACE_NAME, name)
|
|
341
|
+
grouped.setdefault(key, []).append(
|
|
342
|
+
{
|
|
343
|
+
"started_at": started,
|
|
344
|
+
"finished_at": finished,
|
|
345
|
+
"duration": finished - started,
|
|
346
|
+
"arguments": arguments,
|
|
347
|
+
"manifest": manifest,
|
|
348
|
+
"had_load": bool(had_load),
|
|
349
|
+
"events_at_cap": bool(at_cap),
|
|
350
|
+
"host_memory_peak_rss_mb": peak_rss_mb,
|
|
351
|
+
"host_memory_job_peak_rss_mb": job_peak_rss_mb,
|
|
352
|
+
}
|
|
353
|
+
)
|
|
354
|
+
return grouped
|
|
355
|
+
|
|
356
|
+
def orphan_workflow_history(self, workspace, workflow_name):
|
|
357
|
+
"""Detach this (workspace, workflow_name)'s finished runs from cost
|
|
358
|
+
history (#274).
|
|
359
|
+
|
|
360
|
+
Deleting a workflow does not delete the job rows that ran it - those
|
|
361
|
+
stay for `list_jobs`/`get_job` and any other audit trail - but a name
|
|
362
|
+
reused afterwards, in this workspace or a fresh one copied from it,
|
|
363
|
+
must not inherit the old identity's figures. Setting `workflow_name`
|
|
364
|
+
to NULL is enough: `finished_runs()` already excludes rows where it
|
|
365
|
+
is NULL, the same rule that already excludes a run from an inline
|
|
366
|
+
definition.
|
|
367
|
+
"""
|
|
368
|
+
with self._lock, self._connect() as connection:
|
|
369
|
+
connection.execute(
|
|
370
|
+
"UPDATE jobs SET workflow_name = NULL"
|
|
371
|
+
" WHERE workspace = ? AND workflow_name = ?",
|
|
372
|
+
(workspace, workflow_name),
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
def events_for(self, job_id):
|
|
376
|
+
"""A finished job's persisted event tail. [] for a job recorded
|
|
377
|
+
before events were kept, None for a job history has never seen -
|
|
378
|
+
the caller needs to tell 'no events' from 'no such job'."""
|
|
379
|
+
with self._lock, self._connect() as connection:
|
|
380
|
+
row = connection.execute(
|
|
381
|
+
"SELECT events FROM jobs WHERE id = ?", (job_id,)
|
|
382
|
+
).fetchone()
|
|
383
|
+
if row is None:
|
|
384
|
+
return None
|
|
385
|
+
if not row[0]:
|
|
386
|
+
return []
|
|
387
|
+
try:
|
|
388
|
+
return json.loads(row[0])
|
|
389
|
+
except json.JSONDecodeError:
|
|
390
|
+
return []
|
|
391
|
+
|
|
392
|
+
def job_for_file(self, file_name, workspace=None):
|
|
393
|
+
"""The most recent job that actually wrote this output file.
|
|
394
|
+
|
|
395
|
+
LIKE metacharacters are escaped - generated names routinely contain
|
|
396
|
+
'_', which would otherwise match any character and let a similarly
|
|
397
|
+
named later job claim the file.
|
|
398
|
+
|
|
399
|
+
A manifest entry marked 'reused' is a step-cache hit republishing an
|
|
400
|
+
earlier run's files, so it is skipped: attribution belongs to the job
|
|
401
|
+
that wrote the file, not to every later run that reused it.
|
|
402
|
+
|
|
403
|
+
`workspace` narrows the scan to one workspace - two workspaces can
|
|
404
|
+
each produce a file with the same relative name, and without this a
|
|
405
|
+
later job in another workspace could wrongly claim the match.
|
|
406
|
+
"""
|
|
407
|
+
escaped = (
|
|
408
|
+
file_name.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_")
|
|
409
|
+
)
|
|
410
|
+
# Unbounded on purpose: every later fixed-seed rerun republishes the
|
|
411
|
+
# file with 'reused', so a LIMIT would let the writing job fall out of
|
|
412
|
+
# the window after that many reruns and leave the file unattributed.
|
|
413
|
+
# The LIKE filter already restricts the scan to manifests naming it.
|
|
414
|
+
query = (
|
|
415
|
+
"SELECT id, status, manifest FROM jobs WHERE manifest LIKE ? ESCAPE '\\'"
|
|
416
|
+
)
|
|
417
|
+
params = [f"%{escaped}%"]
|
|
418
|
+
if workspace:
|
|
419
|
+
query += " AND workspace = ?"
|
|
420
|
+
params.append(workspace)
|
|
421
|
+
query += " ORDER BY finished_at DESC"
|
|
422
|
+
with self._lock, self._connect() as connection:
|
|
423
|
+
rows = connection.execute(query, params).fetchall()
|
|
424
|
+
for row in rows:
|
|
425
|
+
if self._manifest_wrote(row[2], file_name):
|
|
426
|
+
return {"id": row[0], "status": row[1]}
|
|
427
|
+
return None
|
|
428
|
+
|
|
429
|
+
@staticmethod
|
|
430
|
+
def _manifest_wrote(manifest_text, file_name):
|
|
431
|
+
"""Whether this manifest names the file in an entry it wrote itself.
|
|
432
|
+
|
|
433
|
+
A manifest that will not parse falls back to the LIKE match that
|
|
434
|
+
found it - a row recorded before entries carried 'reused' cannot
|
|
435
|
+
have been a reuse anyway.
|
|
436
|
+
"""
|
|
437
|
+
try:
|
|
438
|
+
manifest = json.loads(manifest_text)
|
|
439
|
+
except (TypeError, ValueError):
|
|
440
|
+
return True
|
|
441
|
+
if not isinstance(manifest, list):
|
|
442
|
+
return True
|
|
443
|
+
# A manifest entry names a file the way the run recorded it - a
|
|
444
|
+
# server-recorded manifest holds names relative to the output
|
|
445
|
+
# directory (_relative_output_names), a directly-run workflow's holds
|
|
446
|
+
# absolute paths. The caller names it relative to the output
|
|
447
|
+
# directory, so match on the tail either way - the same relationship
|
|
448
|
+
# the LIKE substring match relied on
|
|
449
|
+
wanted = file_name.replace(os.sep, "/")
|
|
450
|
+
|
|
451
|
+
def names_file(path):
|
|
452
|
+
normalized = path.replace(os.sep, "/")
|
|
453
|
+
return normalized == wanted or normalized.endswith("/" + wanted)
|
|
454
|
+
|
|
455
|
+
return any(
|
|
456
|
+
not entry.get("reused")
|
|
457
|
+
and any(names_file(path) for path in entry.get("files") or [])
|
|
458
|
+
for entry in manifest
|
|
459
|
+
if isinstance(entry, dict)
|
|
460
|
+
)
|
|
461
|
+
|
|
462
|
+
@staticmethod
|
|
463
|
+
def _to_detail(row):
|
|
464
|
+
def parse(text, fallback):
|
|
465
|
+
try:
|
|
466
|
+
return json.loads(text)
|
|
467
|
+
except (TypeError, ValueError):
|
|
468
|
+
return fallback
|
|
469
|
+
|
|
470
|
+
spec = parse(row[7], {})
|
|
471
|
+
# The persisted tail is capped at MAX_PERSISTED_EVENTS, and
|
|
472
|
+
# get_job_events serves that same tail - so counting it, rather than
|
|
473
|
+
# hardcoding 0, keeps event_count truthful about what a caller who
|
|
474
|
+
# pages through get_job_events will actually see (#289)
|
|
475
|
+
events = parse(row[16], [])
|
|
476
|
+
return {
|
|
477
|
+
"id": row[0],
|
|
478
|
+
"workflow": row[1],
|
|
479
|
+
"status": row[2],
|
|
480
|
+
"created_at": row[3],
|
|
481
|
+
"started_at": row[4],
|
|
482
|
+
"finished_at": row[5],
|
|
483
|
+
"arguments": parse(row[6], {}),
|
|
484
|
+
"spec": spec,
|
|
485
|
+
"manifest": parse(row[8], []),
|
|
486
|
+
"warnings": parse(row[9], []),
|
|
487
|
+
"error": row[10],
|
|
488
|
+
"workspace": row[11] or DEFAULT_WORKSPACE_NAME,
|
|
489
|
+
"workflow_name": row[12],
|
|
490
|
+
"run_id": row[13],
|
|
491
|
+
"run_dir": row[14],
|
|
492
|
+
"run_version": row[17],
|
|
493
|
+
"acknowledged": row[15] or ACK_NONE,
|
|
494
|
+
"acknowledged_cost": (spec or {}).get("acknowledged_cost"),
|
|
495
|
+
"traceback": None,
|
|
496
|
+
"event_count": len(events) if isinstance(events, list) else 0,
|
|
497
|
+
"historical": True,
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
class Job:
|
|
502
|
+
"""One workflow execution request and everything observed about it."""
|
|
503
|
+
|
|
504
|
+
def __init__(self, spec):
|
|
505
|
+
self.id = uuid.uuid4().hex[:12]
|
|
506
|
+
self.spec = spec
|
|
507
|
+
self.workflow_name = spec.get("workflow_name", "unknown")
|
|
508
|
+
self.catalog_name = spec.get("catalog_name")
|
|
509
|
+
self.status = QUEUED
|
|
510
|
+
self.created_at = time.time()
|
|
511
|
+
self.started_at = None
|
|
512
|
+
self.finished_at = None
|
|
513
|
+
self.manifest = []
|
|
514
|
+
# A copy: run-time warnings are appended to this list (see
|
|
515
|
+
# _note_progress) and the spec is what a rerun is built from
|
|
516
|
+
self.warnings = list(spec.get("warnings", []))
|
|
517
|
+
self.error = None
|
|
518
|
+
self.traceback = None
|
|
519
|
+
# Which run this job turned out to be - reported by the worker's
|
|
520
|
+
# run_start event, unknown until then and forever for a job that
|
|
521
|
+
# never got that far
|
|
522
|
+
self.run_id = None
|
|
523
|
+
self.run_dir = None
|
|
524
|
+
self.run_version = None
|
|
525
|
+
# Which form of cost acknowledgement queued this job (#85)
|
|
526
|
+
self.acknowledged = spec.get("acknowledged") or ACK_NONE
|
|
527
|
+
# The worker's own high-water mark for this run, from its final
|
|
528
|
+
# memory_info message - None for a run that never got that far (#243)
|
|
529
|
+
self.host_memory_peak_rss_mb = None
|
|
530
|
+
# This job's own contribution to that process-lifetime figure -
|
|
531
|
+
# growth since the job's first phase boundary, or the job's current
|
|
532
|
+
# rss when it caused no growth (#272). None for a run that never got
|
|
533
|
+
# a memory_info message at all
|
|
534
|
+
self.host_memory_job_peak_rss_mb = None
|
|
535
|
+
self.events = []
|
|
536
|
+
# The running summary a poll reads - see _note_progress. Kept as the
|
|
537
|
+
# events arrive rather than derived from the log on request, because
|
|
538
|
+
# the log is trimmed to its last MAX_PERSISTED_EVENTS and a caller
|
|
539
|
+
# polling a long render should not have to page through it to learn
|
|
540
|
+
# that something moved
|
|
541
|
+
self.last_event_at = None
|
|
542
|
+
self.phase = None
|
|
543
|
+
self.phase_detail = None
|
|
544
|
+
self.phase_started_at = None
|
|
545
|
+
self.step_name = None
|
|
546
|
+
self.parent_step = None
|
|
547
|
+
self.step_index = None
|
|
548
|
+
self.total_steps = None
|
|
549
|
+
self.denoise_step = None
|
|
550
|
+
self.denoise_total_steps = None
|
|
551
|
+
self.condition = threading.Condition()
|
|
552
|
+
|
|
553
|
+
def add_event(self, event):
|
|
554
|
+
with self.condition:
|
|
555
|
+
# `at` is seconds since the job started (since it was created,
|
|
556
|
+
# for the events before that). Phases say what a step is waiting
|
|
557
|
+
# on; only a clock on each event says what it cost - the
|
|
558
|
+
# lead-in from `step_start` to the first `pipeline_step` on a
|
|
559
|
+
# reused pipeline is the number a "slow start" report needs
|
|
560
|
+
since = self.started_at if self.started_at is not None else self.created_at
|
|
561
|
+
self.events.append(
|
|
562
|
+
{"seq": len(self.events), "at": round(time.time() - since, 1), **event}
|
|
563
|
+
)
|
|
564
|
+
self._note_progress(event)
|
|
565
|
+
self.condition.notify_all()
|
|
566
|
+
|
|
567
|
+
def _note_progress(self, event):
|
|
568
|
+
"""Fold one event into the running summary.
|
|
569
|
+
|
|
570
|
+
A single-step generation emits `generating` and then nothing until it
|
|
571
|
+
is done, so 'no new events' is the normal state of a healthy run and
|
|
572
|
+
says nothing about whether it is progressing. What answers that is
|
|
573
|
+
how long it has been that way, and how far into the denoise loop it
|
|
574
|
+
got - both of which are here rather than in the event log.
|
|
575
|
+
"""
|
|
576
|
+
now = time.time()
|
|
577
|
+
self.last_event_at = now
|
|
578
|
+
kind = event.get("event")
|
|
579
|
+
if kind == "phase":
|
|
580
|
+
self.phase = event.get("phase")
|
|
581
|
+
self.phase_detail = event.get("detail")
|
|
582
|
+
self.phase_started_at = now
|
|
583
|
+
elif kind == "pipeline_step":
|
|
584
|
+
self.denoise_step = event.get("step")
|
|
585
|
+
self.denoise_total_steps = event.get("total_steps")
|
|
586
|
+
elif kind == "download_progress":
|
|
587
|
+
# Folded into phase_detail rather than a field of its own - a
|
|
588
|
+
# poller already reads phase_detail for what the loading phase
|
|
589
|
+
# is waiting on, and the next "phase" event (loading ending)
|
|
590
|
+
# overwrites it same as any other detail (#343)
|
|
591
|
+
self.phase_detail = format_progress(
|
|
592
|
+
event.get("repo_id"),
|
|
593
|
+
event.get("downloaded_bytes"),
|
|
594
|
+
event.get("bytes_per_second"),
|
|
595
|
+
event.get("seconds_since_bytes_changed"),
|
|
596
|
+
)
|
|
597
|
+
elif kind == "warning":
|
|
598
|
+
# Both channels, on purpose: the event log keeps the moment it
|
|
599
|
+
# happened, `warnings` keeps it where a caller who polled the
|
|
600
|
+
# finished job will actually look, since a warning about the
|
|
601
|
+
# artifact outlives the run that noticed it (#82). The step it
|
|
602
|
+
# fired in is the run's, not the warning's - the engine warns
|
|
603
|
+
# from inside a step without knowing which one it is.
|
|
604
|
+
#
|
|
605
|
+
# A phase-stall report (#176) is the exception: it is a moment,
|
|
606
|
+
# not a fact about the result - a 90 s cold load says "still in
|
|
607
|
+
# phase 'loading'" three times and then succeeds - so it stays
|
|
608
|
+
# in the event log only. `warnings` is the channel a consumer
|
|
609
|
+
# reads after the run, and the regression suites assert it is
|
|
610
|
+
# empty on a clean one.
|
|
611
|
+
message = event.get("message")
|
|
612
|
+
if message and event.get("kind") != "phase_stall":
|
|
613
|
+
named = f"{self.step_name}: {message}" if self.step_name else message
|
|
614
|
+
if named not in self.warnings:
|
|
615
|
+
self.warnings.append(named)
|
|
616
|
+
elif kind == "step_start":
|
|
617
|
+
self.step_name = event.get("step")
|
|
618
|
+
# A sub-workflow counts its own steps from zero; what a caller
|
|
619
|
+
# watching a composed run needs is where the run it queued has
|
|
620
|
+
# got to, so the parent's counter wins when the event carries
|
|
621
|
+
# one and the step name stays the child's (#90)
|
|
622
|
+
self.parent_step = event.get("parent_step")
|
|
623
|
+
self.step_index = event.get("parent_index", event.get("index"))
|
|
624
|
+
self.total_steps = event.get("parent_total_steps", event.get("total_steps"))
|
|
625
|
+
# A new step's denoise loop has not started; the previous step's
|
|
626
|
+
# count would read as this one's progress
|
|
627
|
+
self.denoise_step = None
|
|
628
|
+
self.denoise_total_steps = None
|
|
629
|
+
|
|
630
|
+
def progress(self):
|
|
631
|
+
"""Where a running job has got to, or None for one that has not
|
|
632
|
+
started - a terminal job has a manifest, which is a better answer
|
|
633
|
+
than a stale phase, except for FAILED: the manifest is only the
|
|
634
|
+
steps that finished, not the one that was running when the job died,
|
|
635
|
+
and that phase (`loading` / `generating` / `decoding` / `saving`) is
|
|
636
|
+
the fastest way to tell what killed it without reading a traceback
|
|
637
|
+
(#269). Frozen at `finished_at` rather than read against the current
|
|
638
|
+
clock, so `seconds_in_phase` reports how long the dead step had been
|
|
639
|
+
running rather than growing forever after the job is long over."""
|
|
640
|
+
if self.last_event_at is None or self.status not in (RUNNING, FAILED):
|
|
641
|
+
return None
|
|
642
|
+
now = (
|
|
643
|
+
self.finished_at
|
|
644
|
+
if self.status == FAILED and self.finished_at
|
|
645
|
+
else time.time()
|
|
646
|
+
)
|
|
647
|
+
summary = {
|
|
648
|
+
"step": self.step_name,
|
|
649
|
+
# The step of the queued workflow the one above is running
|
|
650
|
+
# inside, for a composed run; null when they are the same thing
|
|
651
|
+
"parent_step": self.parent_step,
|
|
652
|
+
"step_index": self.step_index,
|
|
653
|
+
"total_steps": self.total_steps,
|
|
654
|
+
"phase": self.phase,
|
|
655
|
+
"phase_detail": self.phase_detail,
|
|
656
|
+
"seconds_in_phase": (
|
|
657
|
+
round(now - self.phase_started_at, 1) if self.phase_started_at else None
|
|
658
|
+
),
|
|
659
|
+
# The one number that separates a slow run from a hung one -
|
|
660
|
+
# but only once the denoise loop is running, see below
|
|
661
|
+
"seconds_since_event": round(now - self.last_event_at, 1),
|
|
662
|
+
# Always present, null until the loop starts. A key that only
|
|
663
|
+
# appears once there is a count to report cannot be told apart
|
|
664
|
+
# from a key that is missing because nothing is happening: the
|
|
665
|
+
# lead-in to `generating` - encoding the prompt and any
|
|
666
|
+
# reference image or audio - is over a minute of silence on a
|
|
667
|
+
# large video model, and read as an absent counter it looks
|
|
668
|
+
# exactly like a wedged denoise loop. Null here means the loop
|
|
669
|
+
# has not started; a number that stops moving is the stuck one
|
|
670
|
+
"denoise_step": self.denoise_step,
|
|
671
|
+
"denoise_total_steps": self.denoise_total_steps,
|
|
672
|
+
}
|
|
673
|
+
return summary
|
|
674
|
+
|
|
675
|
+
def finish(self, status, error=None, traceback_text=None):
|
|
676
|
+
self.status = status
|
|
677
|
+
self.finished_at = time.time()
|
|
678
|
+
self.error = error
|
|
679
|
+
self.traceback = traceback_text
|
|
680
|
+
self.add_event({"event": "job_status", "status": status})
|
|
681
|
+
|
|
682
|
+
def events_after(self, after_seq):
|
|
683
|
+
# Clamped: an 'after' below -1 would slice from the END of the log
|
|
684
|
+
# (events[-4:] for after=-5) and silently drop the earlier events a
|
|
685
|
+
# client asking for everything expects
|
|
686
|
+
after_seq = max(after_seq, -1)
|
|
687
|
+
with self.condition:
|
|
688
|
+
return self.events[after_seq + 1 :]
|
|
689
|
+
|
|
690
|
+
def wait_for_event(self, after_seq, timeout):
|
|
691
|
+
"""Block until an event past after_seq exists or the job ends."""
|
|
692
|
+
with self.condition:
|
|
693
|
+
if len(self.events) > after_seq + 1 or self.status in TERMINAL_STATES:
|
|
694
|
+
return
|
|
695
|
+
self.condition.wait(timeout)
|
|
696
|
+
|
|
697
|
+
def summary(self):
|
|
698
|
+
return {
|
|
699
|
+
"id": self.id,
|
|
700
|
+
"workflow": self.workflow_name,
|
|
701
|
+
"workflow_name": self.catalog_name,
|
|
702
|
+
"status": self.status,
|
|
703
|
+
"created_at": self.created_at,
|
|
704
|
+
"started_at": self.started_at,
|
|
705
|
+
"finished_at": self.finished_at,
|
|
706
|
+
# Which workspace this job runs in - a live job's spec may not
|
|
707
|
+
# carry one yet (e.g. a caller that never named a workspace),
|
|
708
|
+
# so it defaults the same way history's column does
|
|
709
|
+
"workspace": self.spec.get("workspace") or DEFAULT_WORKSPACE_NAME,
|
|
710
|
+
"run_id": self.run_id,
|
|
711
|
+
# The run's ordinal - 'v4' - so the job that just ran can be
|
|
712
|
+
# named the way the gallery will name it
|
|
713
|
+
"run_version": self.run_version,
|
|
714
|
+
"acknowledged": self.acknowledged,
|
|
715
|
+
}
|
|
716
|
+
|
|
717
|
+
def detail(self):
|
|
718
|
+
return {
|
|
719
|
+
**self.summary(),
|
|
720
|
+
"arguments": self.spec.get("arguments", {}),
|
|
721
|
+
"warnings": self.warnings,
|
|
722
|
+
"manifest": self.manifest,
|
|
723
|
+
"error": self.error,
|
|
724
|
+
"traceback": self.traceback,
|
|
725
|
+
"event_count": len(self.events),
|
|
726
|
+
"run_dir": self.run_dir,
|
|
727
|
+
"acknowledged_cost": self.spec.get("acknowledged_cost"),
|
|
728
|
+
"progress": self.progress(),
|
|
729
|
+
}
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
class JobManager:
|
|
733
|
+
"""Serializes job execution onto the one GPU worker process."""
|
|
734
|
+
|
|
735
|
+
def __init__(
|
|
736
|
+
self,
|
|
737
|
+
output_dir,
|
|
738
|
+
log_level="INFO",
|
|
739
|
+
worker_manager=None,
|
|
740
|
+
history_path=None,
|
|
741
|
+
workflow_dir=None,
|
|
742
|
+
):
|
|
743
|
+
self.output_dir = validate_output_path(output_dir, None)
|
|
744
|
+
# Confines workflow_path/base_dir/sub-workflow resolution for every
|
|
745
|
+
# job this manager submits - the server's configured workflow_dir
|
|
746
|
+
self.workflow_dir = workflow_dir
|
|
747
|
+
os.makedirs(self.output_dir, exist_ok=True)
|
|
748
|
+
self.log_level = log_level
|
|
749
|
+
self.worker_manager = worker_manager or WorkerManager()
|
|
750
|
+
self.history = JobHistory(history_path or resolve_path("jobs.sqlite"))
|
|
751
|
+
self.jobs = {}
|
|
752
|
+
self.last_memory = None
|
|
753
|
+
self.last_memory_at = None
|
|
754
|
+
# Reentrant: cancel() finishes a queued job while holding it, and
|
|
755
|
+
# _finish's terminal-job trim needs it again on the same thread
|
|
756
|
+
self._lock = threading.RLock() # guards job state transitions
|
|
757
|
+
# Pending job ids in run order - a list, not a Queue, so the queue
|
|
758
|
+
# can be reordered while jobs wait
|
|
759
|
+
self._pending = []
|
|
760
|
+
self._wake = threading.Condition(self._lock)
|
|
761
|
+
self._worker_lock = threading.Lock() # guards worker communication
|
|
762
|
+
self._current_job_id = None
|
|
763
|
+
self._stop = threading.Event()
|
|
764
|
+
self._runner = threading.Thread(
|
|
765
|
+
target=self._run_loop, daemon=True, name="job-runner"
|
|
766
|
+
)
|
|
767
|
+
self._runner.start()
|
|
768
|
+
|
|
769
|
+
# ------------------------------------------------------------- submission
|
|
770
|
+
|
|
771
|
+
def submit(
|
|
772
|
+
self,
|
|
773
|
+
workflow_path=None,
|
|
774
|
+
workflow=None,
|
|
775
|
+
arguments=None,
|
|
776
|
+
base_dir=None,
|
|
777
|
+
workflow_dir=None,
|
|
778
|
+
output_dir=None,
|
|
779
|
+
asset_dir=None,
|
|
780
|
+
workspace=None,
|
|
781
|
+
catalog_name=None,
|
|
782
|
+
acknowledged=ACK_NONE,
|
|
783
|
+
acknowledged_cost=None,
|
|
784
|
+
):
|
|
785
|
+
"""Validate a job request and queue it. Raises ValueError on a bad
|
|
786
|
+
request so the HTTP layer can answer 400 before anything runs.
|
|
787
|
+
|
|
788
|
+
`workflow_dir` overrides this job's confinement root for a workflow
|
|
789
|
+
that lives outside the writable directory - an example or a builtin,
|
|
790
|
+
which the caller has already resolved against the search path. The
|
|
791
|
+
worker re-validates against whatever this job records, so the
|
|
792
|
+
override travels with the job rather than widening the manager.
|
|
793
|
+
|
|
794
|
+
`output_dir`, `asset_dir` and `workspace` name which workspace this
|
|
795
|
+
job runs in. They travel with the job for the same reason: one
|
|
796
|
+
server holds several workspaces, and the process-wide roots would
|
|
797
|
+
make every job belong to whichever one was configured at startup.
|
|
798
|
+
|
|
799
|
+
`catalog_name` is the listing name the caller resolved `workflow_path`
|
|
800
|
+
from, kept for history; None for an inline definition.
|
|
801
|
+
|
|
802
|
+
`acknowledged` is the form of cost acknowledgement the caller gave
|
|
803
|
+
(none/boolean/bound) and `acknowledged_cost` the bound object - both
|
|
804
|
+
recorded, neither checked here; the route checks (#85).
|
|
805
|
+
"""
|
|
806
|
+
arguments = arguments or {}
|
|
807
|
+
if (workflow_path is None) == (workflow is None):
|
|
808
|
+
raise ValueError("Provide exactly one of workflow_path or workflow")
|
|
809
|
+
|
|
810
|
+
confinement = workflow_dir or self.workflow_dir
|
|
811
|
+
job_output_dir = (
|
|
812
|
+
validate_output_path(output_dir, None) if output_dir else self.output_dir
|
|
813
|
+
)
|
|
814
|
+
os.makedirs(job_output_dir, exist_ok=True)
|
|
815
|
+
|
|
816
|
+
if workflow_path is not None:
|
|
817
|
+
# Loads and schema-validates now - a bad path or file fails the
|
|
818
|
+
# request, not the queue. Checked against the caller's arguments,
|
|
819
|
+
# not the document alone - a bare validate() checks the document
|
|
820
|
+
# with no arguments and so could refuse a run _candidate_for had
|
|
821
|
+
# already accepted for the same call (#415, the run_workflow
|
|
822
|
+
# mirror of #414)
|
|
823
|
+
loaded = workflow_from_file(workflow_path, job_output_dir, confinement)
|
|
824
|
+
errors = loaded.validation_errors(arguments=arguments)
|
|
825
|
+
if errors:
|
|
826
|
+
raise Exception(format_validation_errors(errors))
|
|
827
|
+
spec = {
|
|
828
|
+
"workflow_path": workflow_path,
|
|
829
|
+
"workflow_name": loaded.name,
|
|
830
|
+
"arguments": arguments,
|
|
831
|
+
"workflow_dir": confinement,
|
|
832
|
+
}
|
|
833
|
+
else:
|
|
834
|
+
# workflow_from_definition validates base_dir - it is HTTP-supplied
|
|
835
|
+
# path input and goes through the security layer like every path
|
|
836
|
+
loaded = workflow_from_definition(
|
|
837
|
+
copy.deepcopy(workflow), job_output_dir, base_dir, confinement
|
|
838
|
+
)
|
|
839
|
+
errors = loaded.validation_errors(arguments=arguments)
|
|
840
|
+
if errors:
|
|
841
|
+
raise Exception(format_validation_errors(errors))
|
|
842
|
+
spec = {
|
|
843
|
+
"workflow": workflow,
|
|
844
|
+
# Must match workflow_from_definition's fallback - the worker
|
|
845
|
+
# re-validates this against workflow_dir
|
|
846
|
+
"base_dir": base_dir
|
|
847
|
+
or (os.path.abspath(confinement) if confinement else os.getcwd()),
|
|
848
|
+
"workflow_name": loaded.name,
|
|
849
|
+
"arguments": arguments,
|
|
850
|
+
# Must be the same root the worker re-validates base_dir
|
|
851
|
+
# against (workflow_from_definition -> validate_path) - this
|
|
852
|
+
# job's own confinement, not the manager's process-wide
|
|
853
|
+
# default, or a named workspace's inline job fails after a
|
|
854
|
+
# 201 the moment base_dir and workflow_dir disagree
|
|
855
|
+
"workflow_dir": confinement,
|
|
856
|
+
}
|
|
857
|
+
|
|
858
|
+
# Which workspace this job runs in, and the roots that follow from
|
|
859
|
+
# it - recorded on the job so history, the worker command and a
|
|
860
|
+
# rerun all agree without re-deriving them
|
|
861
|
+
spec["workspace"] = workspace
|
|
862
|
+
spec["catalog_name"] = catalog_name
|
|
863
|
+
# The acknowledgement form travels with the job so history can say
|
|
864
|
+
# whether this run was consented to at its actual size (#85)
|
|
865
|
+
spec["acknowledged"] = acknowledged
|
|
866
|
+
if acknowledged_cost is not None:
|
|
867
|
+
spec["acknowledged_cost"] = acknowledged_cost
|
|
868
|
+
spec["output_dir"] = job_output_dir
|
|
869
|
+
if asset_dir:
|
|
870
|
+
spec["asset_dir"] = asset_dir
|
|
871
|
+
|
|
872
|
+
# The caller's own arguments, checked against the variables this
|
|
873
|
+
# workflow declares. set_variables makes the same check at the top of
|
|
874
|
+
# the run, so a bad name failed a job that had already been queued -
|
|
875
|
+
# and a workflow declaring no variables dropped every argument in
|
|
876
|
+
# silence. Refused here instead, while it is still a 400
|
|
877
|
+
problems = argument_errors(loaded.workflow_definition, arguments)
|
|
878
|
+
if problems:
|
|
879
|
+
raise ValueError(
|
|
880
|
+
"; ".join(
|
|
881
|
+
f"{problem['path']}: {problem['message']}" for problem in problems
|
|
882
|
+
)
|
|
883
|
+
)
|
|
884
|
+
|
|
885
|
+
# Signature-level check of pipeline arguments - the typo that would
|
|
886
|
+
# otherwise be a TypeError after the model loads becomes a warning
|
|
887
|
+
# the client sees at submission
|
|
888
|
+
spec["warnings"] = workflow_argument_warnings(
|
|
889
|
+
loaded.workflow_definition, arguments
|
|
890
|
+
)
|
|
891
|
+
|
|
892
|
+
job = Job(spec)
|
|
893
|
+
with self._lock:
|
|
894
|
+
self.jobs[job.id] = job
|
|
895
|
+
job.add_event({"event": "job_status", "status": QUEUED})
|
|
896
|
+
with self._wake:
|
|
897
|
+
self._pending.append(job.id)
|
|
898
|
+
self._wake.notify()
|
|
899
|
+
logger.info(f"Queued job {job.id} for workflow {job.workflow_name}")
|
|
900
|
+
return job
|
|
901
|
+
|
|
902
|
+
def get(self, job_id):
|
|
903
|
+
"""A live Job, or a historical detail dict for a finished past run."""
|
|
904
|
+
job = self.jobs.get(job_id)
|
|
905
|
+
if job is not None:
|
|
906
|
+
return job
|
|
907
|
+
return self.history.get(job_id)
|
|
908
|
+
|
|
909
|
+
def definition(self, job_id):
|
|
910
|
+
"""The workflow JSON a job ran, for a read-only view of it.
|
|
911
|
+
|
|
912
|
+
An inline definition comes straight from the spec; a job launched
|
|
913
|
+
from a path is re-read from disk, confined to the root the job ran
|
|
914
|
+
against. None when there is no such job, or when the file it named
|
|
915
|
+
has since moved, grown past the size limit or stopped parsing - a
|
|
916
|
+
graph of the run is a nicety, never a reason to fail the page.
|
|
917
|
+
"""
|
|
918
|
+
job = self.jobs.get(job_id)
|
|
919
|
+
if job is not None:
|
|
920
|
+
spec = job.spec
|
|
921
|
+
else:
|
|
922
|
+
historical = self.history.get(job_id)
|
|
923
|
+
if historical is None:
|
|
924
|
+
return None
|
|
925
|
+
spec = historical.get("spec") or {}
|
|
926
|
+
inline = spec.get("workflow")
|
|
927
|
+
if inline is not None:
|
|
928
|
+
return copy.deepcopy(inline)
|
|
929
|
+
path = spec.get("workflow_path")
|
|
930
|
+
if not path:
|
|
931
|
+
return None
|
|
932
|
+
try:
|
|
933
|
+
validated = validate_workflow_path(
|
|
934
|
+
path, spec.get("workflow_dir") or self.workflow_dir
|
|
935
|
+
)
|
|
936
|
+
validate_json_size(validated)
|
|
937
|
+
with open(validated, "r") as file:
|
|
938
|
+
return json.load(file)
|
|
939
|
+
except (SecurityError, OSError, ValueError):
|
|
940
|
+
logger.debug(f"No workflow definition available for job {job_id}")
|
|
941
|
+
return None
|
|
942
|
+
|
|
943
|
+
def realized(self, job_id):
|
|
944
|
+
"""The realized workflow a job ran, or None when the job predates
|
|
945
|
+
run tracking or its run directory no longer holds the file.
|
|
946
|
+
|
|
947
|
+
Read from the job's own output directory, not the manager's: one
|
|
948
|
+
server holds several workspaces, and a job carries the root it ran
|
|
949
|
+
against. The join is confined to that root, so a run_dir read back
|
|
950
|
+
out of the database cannot name anything outside it.
|
|
951
|
+
"""
|
|
952
|
+
job = self.jobs.get(job_id)
|
|
953
|
+
if job is not None:
|
|
954
|
+
run_dir = job.run_dir
|
|
955
|
+
output_dir = job.spec.get("output_dir") or self.output_dir
|
|
956
|
+
else:
|
|
957
|
+
historical = self.history.get(job_id)
|
|
958
|
+
if historical is None:
|
|
959
|
+
return None
|
|
960
|
+
run_dir = historical.get("run_dir")
|
|
961
|
+
output_dir = (historical.get("spec") or {}).get(
|
|
962
|
+
"output_dir"
|
|
963
|
+
) or self.output_dir
|
|
964
|
+
if not run_dir:
|
|
965
|
+
return None
|
|
966
|
+
try:
|
|
967
|
+
root = validate_output_path(output_dir, None)
|
|
968
|
+
path = validate_path(os.path.join(root, run_dir, REALIZED_FILE_NAME), root)
|
|
969
|
+
validate_json_size(path)
|
|
970
|
+
with open(path, "r") as file:
|
|
971
|
+
return json.load(file)
|
|
972
|
+
except (SecurityError, OSError, ValueError) as e:
|
|
973
|
+
logger.debug(f"No realized workflow for job {job_id}: {e}")
|
|
974
|
+
return None
|
|
975
|
+
|
|
976
|
+
def seed_variable(self, job_id):
|
|
977
|
+
"""The variable this job's workflow draws its seed from, or None.
|
|
978
|
+
|
|
979
|
+
Read from the workflow as written, never from the realized copy the
|
|
980
|
+
run wrote: realization pins the top-level seed to the integer the run
|
|
981
|
+
used, so a realized workflow always looks like it names a literal.
|
|
982
|
+
|
|
983
|
+
None means a new-seed rerun has nowhere to put one - either the seed
|
|
984
|
+
is a literal (an argument cannot override it) or the workflow names
|
|
985
|
+
no seed at all, in which case every run already draws a fresh one and
|
|
986
|
+
the step cache is off.
|
|
987
|
+
"""
|
|
988
|
+
definition = self.definition(job_id)
|
|
989
|
+
seed = (definition or {}).get("seed")
|
|
990
|
+
if not isinstance(seed, str) or not seed.startswith(VARIABLE_PREFIX):
|
|
991
|
+
return None
|
|
992
|
+
name = seed.removeprefix(VARIABLE_PREFIX)
|
|
993
|
+
return name if name in (definition.get("variables") or {}) else None
|
|
994
|
+
|
|
995
|
+
def rerun_spec(self, job_id):
|
|
996
|
+
"""The spec and arguments a rerun of `job_id` would submit, as
|
|
997
|
+
(spec, arguments), or None for an unknown job - split from rerun()
|
|
998
|
+
so a route can plan the run before queuing it (#85)."""
|
|
999
|
+
job = self.jobs.get(job_id)
|
|
1000
|
+
if job is not None:
|
|
1001
|
+
spec = {key: job.spec[key] for key in RERUN_SPEC_KEYS if key in job.spec}
|
|
1002
|
+
return spec, job.spec.get("arguments", {})
|
|
1003
|
+
historical = self.history.get(job_id)
|
|
1004
|
+
if historical is None:
|
|
1005
|
+
return None
|
|
1006
|
+
spec = {
|
|
1007
|
+
key: historical["spec"][key]
|
|
1008
|
+
for key in RERUN_SPEC_KEYS
|
|
1009
|
+
if key in historical["spec"]
|
|
1010
|
+
}
|
|
1011
|
+
return spec, historical["arguments"]
|
|
1012
|
+
|
|
1013
|
+
def rerun(
|
|
1014
|
+
self, job_id, new_seed=False, acknowledged=ACK_NONE, acknowledged_cost=None
|
|
1015
|
+
):
|
|
1016
|
+
"""Queue a fresh job from a previous job's spec.
|
|
1017
|
+
|
|
1018
|
+
Every root the original ran against (workflow_dir/output_dir/
|
|
1019
|
+
asset_dir/workspace) rides along, not just the workflow identity -
|
|
1020
|
+
otherwise a rerun of a job from a named workspace would fall back to
|
|
1021
|
+
the manager's process-wide default and silently run somewhere else.
|
|
1022
|
+
|
|
1023
|
+
`new_seed` draws a fresh seed into the workflow's seed variable. A
|
|
1024
|
+
plain rerun of a seeded workflow repeats its arguments exactly, which
|
|
1025
|
+
makes every step a step-cache hit: it republishes the earlier run's
|
|
1026
|
+
files in a fraction of a second and generates nothing. That is the
|
|
1027
|
+
cache doing its job - the same seed and the same inputs would produce
|
|
1028
|
+
the same pixels - so the way to actually get another image is to
|
|
1029
|
+
change the seed, and this is that.
|
|
1030
|
+
|
|
1031
|
+
`acknowledged` and `acknowledged_cost` are this request's own; the
|
|
1032
|
+
original's bound object rides along in the spec for the record when
|
|
1033
|
+
the request brought none.
|
|
1034
|
+
"""
|
|
1035
|
+
prepared = self.rerun_spec(job_id)
|
|
1036
|
+
if prepared is None:
|
|
1037
|
+
return None
|
|
1038
|
+
spec, arguments = prepared
|
|
1039
|
+
|
|
1040
|
+
if new_seed:
|
|
1041
|
+
variable = self.seed_variable(job_id)
|
|
1042
|
+
if variable is None:
|
|
1043
|
+
raise ValueError(
|
|
1044
|
+
"This workflow does not draw its seed from a variable, so "
|
|
1045
|
+
"a rerun cannot change it. A workflow with no seed at all "
|
|
1046
|
+
"already draws a fresh one every run."
|
|
1047
|
+
)
|
|
1048
|
+
# Bounded so the number survives its trip through a browser as
|
|
1049
|
+
# JSON - see SEED_BITS
|
|
1050
|
+
arguments = {**arguments, variable: secrets.randbits(SEED_BITS)}
|
|
1051
|
+
|
|
1052
|
+
workspace = spec.get("workspace")
|
|
1053
|
+
if (
|
|
1054
|
+
workspace
|
|
1055
|
+
and workspace != DEFAULT_WORKSPACE_NAME
|
|
1056
|
+
and spec.get("output_dir")
|
|
1057
|
+
and not os.path.isdir(spec["output_dir"])
|
|
1058
|
+
):
|
|
1059
|
+
raise ValueError(f"Workspace '{workspace}' the job ran in no longer exists")
|
|
1060
|
+
|
|
1061
|
+
return self.submit(
|
|
1062
|
+
workflow_path=spec.get("workflow_path"),
|
|
1063
|
+
workflow=spec.get("workflow"),
|
|
1064
|
+
arguments=arguments,
|
|
1065
|
+
base_dir=spec.get("base_dir"),
|
|
1066
|
+
workflow_dir=spec.get("workflow_dir"),
|
|
1067
|
+
output_dir=spec.get("output_dir"),
|
|
1068
|
+
asset_dir=spec.get("asset_dir"),
|
|
1069
|
+
workspace=workspace,
|
|
1070
|
+
catalog_name=spec.get("catalog_name"),
|
|
1071
|
+
acknowledged=acknowledged,
|
|
1072
|
+
acknowledged_cost=(
|
|
1073
|
+
acknowledged_cost
|
|
1074
|
+
if acknowledged_cost is not None
|
|
1075
|
+
else spec.get("acknowledged_cost")
|
|
1076
|
+
),
|
|
1077
|
+
)
|
|
1078
|
+
|
|
1079
|
+
def queue_position(self, job_id):
|
|
1080
|
+
"""Index in the waiting queue, or None when the job is not queued."""
|
|
1081
|
+
with self._lock:
|
|
1082
|
+
return self._pending.index(job_id) if job_id in self._pending else None
|
|
1083
|
+
|
|
1084
|
+
def describe(self, job):
|
|
1085
|
+
"""A live job's detail plus its queue position while it waits - what
|
|
1086
|
+
the per-job endpoints return, so a client holding one job can say
|
|
1087
|
+
where it stands without fetching the whole list."""
|
|
1088
|
+
detail = job.detail()
|
|
1089
|
+
position = self.queue_position(job.id)
|
|
1090
|
+
if position is not None:
|
|
1091
|
+
detail["queue_position"] = position
|
|
1092
|
+
return detail
|
|
1093
|
+
|
|
1094
|
+
def list(self, workspace=None, statuses=None):
|
|
1095
|
+
"""All jobs, live and historical, sorted by creation. `workspace` filters to one workspace; omitted, the list
|
|
1096
|
+
spans every workspace the server holds, unchanged from before
|
|
1097
|
+
workspaces existed. `statuses` filters to a set of job states
|
|
1098
|
+
('queued', 'running', 'succeeded', 'failed', 'cancelled'); omitted,
|
|
1099
|
+
every state is listed."""
|
|
1100
|
+
statuses = set(statuses) if statuses else None
|
|
1101
|
+
with self._lock:
|
|
1102
|
+
live = sorted(self.jobs.values(), key=lambda j: j.created_at)
|
|
1103
|
+
positions = {job_id: i for i, job_id in enumerate(self._pending)}
|
|
1104
|
+
live_ids = {job.id for job in live}
|
|
1105
|
+
summaries = []
|
|
1106
|
+
for job in live:
|
|
1107
|
+
summary = job.summary()
|
|
1108
|
+
if workspace and summary["workspace"] != workspace:
|
|
1109
|
+
continue
|
|
1110
|
+
if statuses and summary["status"] not in statuses:
|
|
1111
|
+
continue
|
|
1112
|
+
if job.id in positions:
|
|
1113
|
+
summary["queue_position"] = positions[job.id]
|
|
1114
|
+
summaries.append(summary)
|
|
1115
|
+
for historical in self.history.recent_summaries(
|
|
1116
|
+
workspace=workspace, statuses=statuses
|
|
1117
|
+
):
|
|
1118
|
+
if historical["id"] not in live_ids:
|
|
1119
|
+
summaries.append(historical)
|
|
1120
|
+
summaries.sort(key=lambda summary: summary["created_at"] or 0)
|
|
1121
|
+
return summaries
|
|
1122
|
+
|
|
1123
|
+
# ------------------------------------------------------------ cancel/stop
|
|
1124
|
+
|
|
1125
|
+
def cancel(self, job_id):
|
|
1126
|
+
"""Cancel a queued or running job. Returns the job's status after the
|
|
1127
|
+
request, or None for an unknown job."""
|
|
1128
|
+
job = self.jobs.get(job_id)
|
|
1129
|
+
if job is None:
|
|
1130
|
+
return None
|
|
1131
|
+
with self._lock:
|
|
1132
|
+
if job.status in TERMINAL_STATES:
|
|
1133
|
+
return job.status
|
|
1134
|
+
if job.status == QUEUED:
|
|
1135
|
+
if job.id in self._pending:
|
|
1136
|
+
self._pending.remove(job.id)
|
|
1137
|
+
self._finish(job, CANCELLED)
|
|
1138
|
+
return job.status
|
|
1139
|
+
if job.status == RUNNING and self._current_job_id == job.id:
|
|
1140
|
+
try:
|
|
1141
|
+
self.worker_manager.cancel()
|
|
1142
|
+
except Exception as e:
|
|
1143
|
+
logger.warning(f"Could not send cancel for job {job_id}: {e}")
|
|
1144
|
+
return job.status
|
|
1145
|
+
|
|
1146
|
+
def move(self, job_id, direction):
|
|
1147
|
+
"""Reorder a queued job: 'up'/'down' swap with a neighbour,
|
|
1148
|
+
'front'/'back' go to the ends. Returns the new pending order, or
|
|
1149
|
+
None for a job that is not queued (finished, running, unknown)."""
|
|
1150
|
+
if direction not in ("up", "down", "front", "back"):
|
|
1151
|
+
raise ValueError(f"Unknown queue direction '{direction}'")
|
|
1152
|
+
with self._lock:
|
|
1153
|
+
if job_id not in self._pending:
|
|
1154
|
+
return None
|
|
1155
|
+
index = self._pending.index(job_id)
|
|
1156
|
+
self._pending.pop(index)
|
|
1157
|
+
if direction == "front":
|
|
1158
|
+
index = 0
|
|
1159
|
+
elif direction == "back":
|
|
1160
|
+
index = len(self._pending)
|
|
1161
|
+
elif direction == "up":
|
|
1162
|
+
index = max(0, index - 1)
|
|
1163
|
+
else:
|
|
1164
|
+
index = min(len(self._pending), index + 1)
|
|
1165
|
+
self._pending.insert(index, job_id)
|
|
1166
|
+
return list(self._pending)
|
|
1167
|
+
|
|
1168
|
+
def shutdown(self):
|
|
1169
|
+
self._stop.set()
|
|
1170
|
+
with self._wake:
|
|
1171
|
+
self._wake.notify_all()
|
|
1172
|
+
self._runner.join(timeout=5)
|
|
1173
|
+
self.worker_manager.shutdown_worker()
|
|
1174
|
+
|
|
1175
|
+
# ---------------------------------------------------------------- runner
|
|
1176
|
+
|
|
1177
|
+
def _run_loop(self):
|
|
1178
|
+
while not self._stop.is_set():
|
|
1179
|
+
with self._wake:
|
|
1180
|
+
while not self._pending and not self._stop.is_set():
|
|
1181
|
+
self._wake.wait()
|
|
1182
|
+
if self._stop.is_set():
|
|
1183
|
+
return
|
|
1184
|
+
job_id = self._pending.pop(0)
|
|
1185
|
+
job = self.jobs.get(job_id)
|
|
1186
|
+
if job is None or job.status != QUEUED:
|
|
1187
|
+
continue # cancelled while waiting
|
|
1188
|
+
self._run_job(job)
|
|
1189
|
+
|
|
1190
|
+
def _finish(self, job, status, error=None, traceback_text=None):
|
|
1191
|
+
job.finish(status, error=error, traceback_text=traceback_text)
|
|
1192
|
+
try:
|
|
1193
|
+
self.history.record(job)
|
|
1194
|
+
except Exception as e:
|
|
1195
|
+
logger.warning(f"Could not persist job {job.id}: {e}")
|
|
1196
|
+
self._trim_terminal_jobs()
|
|
1197
|
+
|
|
1198
|
+
def _trim_terminal_jobs(self):
|
|
1199
|
+
"""Drop the oldest finished jobs from memory - history has them, and
|
|
1200
|
+
get()/list() fall through to it. Recent ones stay for event replay."""
|
|
1201
|
+
with self._lock:
|
|
1202
|
+
terminal = [
|
|
1203
|
+
job
|
|
1204
|
+
for job in sorted(self.jobs.values(), key=lambda j: j.created_at)
|
|
1205
|
+
if job.status in TERMINAL_STATES
|
|
1206
|
+
]
|
|
1207
|
+
for job in terminal[:-TERMINAL_JOBS_KEPT]:
|
|
1208
|
+
del self.jobs[job.id]
|
|
1209
|
+
|
|
1210
|
+
def _run_job(self, job):
|
|
1211
|
+
with self._worker_lock:
|
|
1212
|
+
with self._lock:
|
|
1213
|
+
if job.status != QUEUED:
|
|
1214
|
+
return
|
|
1215
|
+
job.status = RUNNING
|
|
1216
|
+
job.started_at = time.time()
|
|
1217
|
+
self._current_job_id = job.id
|
|
1218
|
+
job.add_event({"event": "job_status", "status": RUNNING})
|
|
1219
|
+
try:
|
|
1220
|
+
self.worker_manager.ensure_worker(self.log_level)
|
|
1221
|
+
command = {
|
|
1222
|
+
"type": "execute",
|
|
1223
|
+
"arguments": job.spec["arguments"],
|
|
1224
|
+
# The job's own roots, so a job queued for one workspace
|
|
1225
|
+
# still runs in it after the manager has served another
|
|
1226
|
+
"output_dir": job.spec.get("output_dir") or self.output_dir,
|
|
1227
|
+
"log_level": self.log_level,
|
|
1228
|
+
}
|
|
1229
|
+
if job.spec.get("asset_dir"):
|
|
1230
|
+
command["asset_dir"] = job.spec["asset_dir"]
|
|
1231
|
+
if "workflow_path" in job.spec:
|
|
1232
|
+
command["workflow_path"] = job.spec["workflow_path"]
|
|
1233
|
+
else:
|
|
1234
|
+
command["workflow"] = job.spec["workflow"]
|
|
1235
|
+
command["base_dir"] = job.spec["base_dir"]
|
|
1236
|
+
command["workflow_dir"] = job.spec.get("workflow_dir")
|
|
1237
|
+
self.worker_manager.send_command(command)
|
|
1238
|
+
outcome = self._consume_results(job)
|
|
1239
|
+
except Exception as e:
|
|
1240
|
+
logger.error(f"Job {job.id} failed: {e}", exc_info=True)
|
|
1241
|
+
outcome = (FAILED, str(e), None)
|
|
1242
|
+
finally:
|
|
1243
|
+
# Cleared BEFORE the terminal status becomes visible - a
|
|
1244
|
+
# client seeing "succeeded" must find the manager idle
|
|
1245
|
+
with self._lock:
|
|
1246
|
+
self._current_job_id = None
|
|
1247
|
+
if job.status not in TERMINAL_STATES:
|
|
1248
|
+
status, error, traceback_text = outcome
|
|
1249
|
+
self._finish(job, status, error=error, traceback_text=traceback_text)
|
|
1250
|
+
|
|
1251
|
+
def _record_manifest(self, job, message):
|
|
1252
|
+
"""What the run wrote, named the way clients address outputs.
|
|
1253
|
+
|
|
1254
|
+
Recorded for a failed or cancelled run as well as a successful one -
|
|
1255
|
+
the files the steps before the stop wrote are on disk either way,
|
|
1256
|
+
and a manifest that omits them is the difference between "this run
|
|
1257
|
+
produced nothing" and "this run produced four of five shots"
|
|
1258
|
+
(T015)."""
|
|
1259
|
+
job.manifest = self._relative_manifest(
|
|
1260
|
+
message.get("manifest", []), job.spec.get("output_dir")
|
|
1261
|
+
)
|
|
1262
|
+
|
|
1263
|
+
def _relative_manifest(self, manifest, output_dir=None):
|
|
1264
|
+
"""A manifest list with every entry's 'files' relativised - the
|
|
1265
|
+
rendering `get_job` and `step_end`/`workflow_end` events must all
|
|
1266
|
+
agree on (#284)."""
|
|
1267
|
+
return [
|
|
1268
|
+
(
|
|
1269
|
+
{
|
|
1270
|
+
**entry,
|
|
1271
|
+
"files": self._relative_output_names(entry["files"], output_dir),
|
|
1272
|
+
}
|
|
1273
|
+
if "files" in entry
|
|
1274
|
+
else entry
|
|
1275
|
+
)
|
|
1276
|
+
for entry in manifest
|
|
1277
|
+
]
|
|
1278
|
+
|
|
1279
|
+
def _relative_output_names(self, paths, output_dir=None):
|
|
1280
|
+
"""The worker reports absolute paths; clients build '/outputs/<name>'
|
|
1281
|
+
URLs, and a run writes under '<output_dir>/<identity>/<run id>/'
|
|
1282
|
+
(dw/workflow.py's effective_output_dir) - so every file is reported
|
|
1283
|
+
by its name relative to the output directory of the job that wrote
|
|
1284
|
+
it, with forward slashes. A path outside it (a task step writing
|
|
1285
|
+
elsewhere) is left as it came.
|
|
1286
|
+
|
|
1287
|
+
The job's own directory, not the manager's: a job in a named
|
|
1288
|
+
workspace writes under that workspace, and naming it relative to the
|
|
1289
|
+
default workspace would produce '../<name>/outputs/...' - a path, not
|
|
1290
|
+
a name."""
|
|
1291
|
+
root = output_dir or self.output_dir
|
|
1292
|
+
names = []
|
|
1293
|
+
for path in paths:
|
|
1294
|
+
relative = os.path.relpath(path, root)
|
|
1295
|
+
if relative.startswith(".."):
|
|
1296
|
+
names.append(path)
|
|
1297
|
+
else:
|
|
1298
|
+
names.append(relative.replace(os.sep, "/"))
|
|
1299
|
+
return names
|
|
1300
|
+
|
|
1301
|
+
def _consume_results(self, job):
|
|
1302
|
+
"""Read worker messages until the run ends; returns the terminal
|
|
1303
|
+
(status, error, traceback) for _run_job to apply once the manager
|
|
1304
|
+
no longer counts the job as current."""
|
|
1305
|
+
while True:
|
|
1306
|
+
try:
|
|
1307
|
+
message = self.worker_manager.get_result()
|
|
1308
|
+
except RuntimeError as e:
|
|
1309
|
+
# The worker died without managing to send anything - a
|
|
1310
|
+
# signal, not an exception, so worker_main's handler never
|
|
1311
|
+
# ran and there is no traceback to be had. The exit code is
|
|
1312
|
+
# the only diagnosis available, and marking the crash here
|
|
1313
|
+
# matters beyond this job: the manager would otherwise go on
|
|
1314
|
+
# believing a dead process is active, and every later call
|
|
1315
|
+
# that talks to it (memory_status above all) would fail
|
|
1316
|
+
# against a queue nobody is reading
|
|
1317
|
+
detail = self.worker_manager.crash_details()
|
|
1318
|
+
self.worker_manager.mark_crashed()
|
|
1319
|
+
reason = f"Worker process died: {detail or e}"
|
|
1320
|
+
logger.error(f"Job {job.id}: {reason}")
|
|
1321
|
+
return (FAILED, reason, None)
|
|
1322
|
+
message_type = message.get("type")
|
|
1323
|
+
|
|
1324
|
+
if message_type == "progress":
|
|
1325
|
+
event = {k: v for k, v in message.items() if k != "type"}
|
|
1326
|
+
if "files" in event:
|
|
1327
|
+
event["files"] = self._relative_output_names(
|
|
1328
|
+
event["files"], job.spec.get("output_dir")
|
|
1329
|
+
)
|
|
1330
|
+
if "manifest" in event:
|
|
1331
|
+
# workflow_end carries the run's full manifest nested
|
|
1332
|
+
# under this key - it must match get_job's rendering of
|
|
1333
|
+
# the same list rather than leaking absolute paths (#284)
|
|
1334
|
+
event["manifest"] = self._relative_manifest(
|
|
1335
|
+
event["manifest"], job.spec.get("output_dir")
|
|
1336
|
+
)
|
|
1337
|
+
if event.get("event") == "run_start":
|
|
1338
|
+
job.run_id = event.get("run_id")
|
|
1339
|
+
job.run_dir = event.get("run_dir")
|
|
1340
|
+
job.run_version = event.get("version")
|
|
1341
|
+
job.add_event(event)
|
|
1342
|
+
elif message_type in ("output", "workflow_loaded"):
|
|
1343
|
+
text = message.get("message") or message.get("workflow_name", "")
|
|
1344
|
+
job.add_event({"event": "log", "message": text})
|
|
1345
|
+
elif message_type == "memory_info":
|
|
1346
|
+
# One per phase boundary now, not just once post-run (#273) -
|
|
1347
|
+
# each folds into the cached reading memory_status() answers
|
|
1348
|
+
# from while the job is busy, which is what makes that call
|
|
1349
|
+
# fresh instead of a refusal for the run's whole duration
|
|
1350
|
+
self._record_memory(message.get("info"))
|
|
1351
|
+
job.add_event({"event": "memory", "info": self.last_memory})
|
|
1352
|
+
# The worker's own high-water mark, latest reading wins (it
|
|
1353
|
+
# is monotonic for the process' life) - persisted as a real
|
|
1354
|
+
# column rather than only inside the trimmed event tail (#243)
|
|
1355
|
+
info = message.get("info") or {}
|
|
1356
|
+
peak = info.get("host_memory_peak_rss_mb")
|
|
1357
|
+
if peak is not None:
|
|
1358
|
+
job.host_memory_peak_rss_mb = peak
|
|
1359
|
+
# This job's own contribution to that process-lifetime peak,
|
|
1360
|
+
# computed against its own baseline (#272) - max() is
|
|
1361
|
+
# defensive; by construction each reading only grows
|
|
1362
|
+
job_peak = info.get("host_memory_job_peak_rss_mb")
|
|
1363
|
+
if job_peak is not None:
|
|
1364
|
+
job.host_memory_job_peak_rss_mb = max(
|
|
1365
|
+
job_peak, job.host_memory_job_peak_rss_mb or 0
|
|
1366
|
+
)
|
|
1367
|
+
elif message_type == "success":
|
|
1368
|
+
self._record_manifest(job, message)
|
|
1369
|
+
return (SUCCEEDED, None, None)
|
|
1370
|
+
elif message_type == "cancelled":
|
|
1371
|
+
self._record_manifest(job, message)
|
|
1372
|
+
return (CANCELLED, None, None)
|
|
1373
|
+
elif message_type == "error":
|
|
1374
|
+
# A failed run's steps too: the ones before the failure wrote
|
|
1375
|
+
# real files, and a job that reports an empty manifest hides
|
|
1376
|
+
# them behind the error that stopped the run
|
|
1377
|
+
self._record_manifest(job, message)
|
|
1378
|
+
return (
|
|
1379
|
+
FAILED,
|
|
1380
|
+
message.get("message"),
|
|
1381
|
+
message.get("traceback"),
|
|
1382
|
+
)
|
|
1383
|
+
elif message_type == "worker_crashed":
|
|
1384
|
+
self.worker_manager.mark_crashed()
|
|
1385
|
+
return (
|
|
1386
|
+
FAILED,
|
|
1387
|
+
f"Worker crashed: {message.get('message')}",
|
|
1388
|
+
message.get("traceback"),
|
|
1389
|
+
)
|
|
1390
|
+
else:
|
|
1391
|
+
logger.warning(f"Unknown worker message type: {message_type}")
|
|
1392
|
+
|
|
1393
|
+
def is_busy(self):
|
|
1394
|
+
"""True while a job is running or queued - the window in which the
|
|
1395
|
+
worker may be reading model files a cache delete would rip out."""
|
|
1396
|
+
with self._lock:
|
|
1397
|
+
if self._current_job_id is not None:
|
|
1398
|
+
return True
|
|
1399
|
+
return any(job.status == QUEUED for job in self.jobs.values())
|
|
1400
|
+
|
|
1401
|
+
def restart_worker_if_idle(self):
|
|
1402
|
+
"""Shut the idle worker down so its next start picks up upgraded
|
|
1403
|
+
imports; the next job respawns it via ensure_worker. Returns False
|
|
1404
|
+
without touching a busy worker - a run in flight keeps the version
|
|
1405
|
+
it started with."""
|
|
1406
|
+
if self.is_busy():
|
|
1407
|
+
return False
|
|
1408
|
+
if not self._worker_lock.acquire(timeout=2):
|
|
1409
|
+
return False
|
|
1410
|
+
try:
|
|
1411
|
+
self.worker_manager.shutdown_worker()
|
|
1412
|
+
return True
|
|
1413
|
+
finally:
|
|
1414
|
+
self._worker_lock.release()
|
|
1415
|
+
|
|
1416
|
+
# ---------------------------------------------------------------- memory
|
|
1417
|
+
|
|
1418
|
+
def _record_memory(self, info):
|
|
1419
|
+
"""Remember a reading and when it was taken, so a later cached answer
|
|
1420
|
+
can say how old it is."""
|
|
1421
|
+
self.last_memory = info
|
|
1422
|
+
self.last_memory_at = time.time() if info is not None else None
|
|
1423
|
+
|
|
1424
|
+
def _cached_memory(self, reason):
|
|
1425
|
+
"""The last reading, labelled with why it is not a live one. A caller
|
|
1426
|
+
comparing two readings must compare only `live: true` ones - a cached
|
|
1427
|
+
`info` was taken at another moment, and while a job loads a model it
|
|
1428
|
+
understates what is resident by however much has loaded since."""
|
|
1429
|
+
info = self.last_memory
|
|
1430
|
+
age = None
|
|
1431
|
+
if info is not None and self.last_memory_at is not None:
|
|
1432
|
+
age = round(time.time() - self.last_memory_at, 1)
|
|
1433
|
+
return {
|
|
1434
|
+
"live": False,
|
|
1435
|
+
"info": info,
|
|
1436
|
+
"stale": info is not None,
|
|
1437
|
+
"reason": reason,
|
|
1438
|
+
"age_seconds": age,
|
|
1439
|
+
}
|
|
1440
|
+
|
|
1441
|
+
def probe_cache(self, command, timeout=5):
|
|
1442
|
+
"""Which steps the worker's step cache would serve for `command` (the
|
|
1443
|
+
fields an execute command carries, minus its type), or None when the
|
|
1444
|
+
answer cannot be had right now - a job is running, the worker is
|
|
1445
|
+
busy, or it did not answer in time. Never blocks a request behind a
|
|
1446
|
+
running job, for the same reason memory_status does not.
|
|
1447
|
+
|
|
1448
|
+
No worker running is a definite answer, not an unknown one: the
|
|
1449
|
+
cache lives in the worker process, so a worker that is not running
|
|
1450
|
+
holds nothing.
|
|
1451
|
+
"""
|
|
1452
|
+
if self._current_job_id is not None:
|
|
1453
|
+
return None
|
|
1454
|
+
if not self.worker_manager.worker_active:
|
|
1455
|
+
return []
|
|
1456
|
+
if not self._worker_lock.acquire(timeout=2):
|
|
1457
|
+
return None
|
|
1458
|
+
# A probe that timed out still answers eventually, onto the same
|
|
1459
|
+
# queue the next probe reads - so each carries an id and a reader
|
|
1460
|
+
# discards every reply that is not its own, rather than reporting
|
|
1461
|
+
# the previous workflow's hit list as this plan's
|
|
1462
|
+
probe_id = uuid.uuid4().hex
|
|
1463
|
+
deadline = time.monotonic() + timeout
|
|
1464
|
+
try:
|
|
1465
|
+
self.worker_manager.send_command(
|
|
1466
|
+
{"type": "probe_cache", "probe_id": probe_id, **command}
|
|
1467
|
+
)
|
|
1468
|
+
while True:
|
|
1469
|
+
remaining = deadline - time.monotonic()
|
|
1470
|
+
if remaining <= 0:
|
|
1471
|
+
return None
|
|
1472
|
+
result = self.worker_manager.get_result(timeout=remaining)
|
|
1473
|
+
if (
|
|
1474
|
+
result.get("type") == "probe_cache"
|
|
1475
|
+
and result.get("probe_id") == probe_id
|
|
1476
|
+
):
|
|
1477
|
+
break
|
|
1478
|
+
logger.debug(f"Discarding a stale worker message: {result.get('type')}")
|
|
1479
|
+
except (RuntimeError, queue.Empty) as e:
|
|
1480
|
+
logger.debug(f"Worker did not answer the cache probe: {e}")
|
|
1481
|
+
return None
|
|
1482
|
+
finally:
|
|
1483
|
+
self._worker_lock.release()
|
|
1484
|
+
cached = result.get("cached")
|
|
1485
|
+
return list(cached) if isinstance(cached, list) else None
|
|
1486
|
+
|
|
1487
|
+
def memory_status(self, timeout=5):
|
|
1488
|
+
"""Live memory stats when the worker is idle; the run's last report
|
|
1489
|
+
while it is busy. The lock acquire is bounded: the runner holds
|
|
1490
|
+
_worker_lock for a job's whole duration, and a poll that raced a job
|
|
1491
|
+
start must fall back to the cached reading, not block for hours.
|
|
1492
|
+
|
|
1493
|
+
`live` says whether `info` was measured by this call. `stale` and
|
|
1494
|
+
`reason` say why it was not, and `age_seconds` how old the cached
|
|
1495
|
+
reading is; `info` is null when there has never been a reading, which
|
|
1496
|
+
means nothing is resident rather than that the answer is unknown."""
|
|
1497
|
+
if self._current_job_id is not None:
|
|
1498
|
+
return self._cached_memory("job_running")
|
|
1499
|
+
if not self.worker_manager.worker_active:
|
|
1500
|
+
return self._cached_memory("worker_stopped")
|
|
1501
|
+
if not self._worker_lock.acquire(timeout=2):
|
|
1502
|
+
return self._cached_memory("worker_busy")
|
|
1503
|
+
try:
|
|
1504
|
+
self.worker_manager.send_command({"type": "memory_status"})
|
|
1505
|
+
result = self.worker_manager.get_result(timeout=timeout)
|
|
1506
|
+
except (RuntimeError, queue.Empty) as e:
|
|
1507
|
+
# A dead worker is exactly when the last reading taken before it
|
|
1508
|
+
# died is worth the most, so report that rather than failing the
|
|
1509
|
+
# request. Without this the caller gets a 503 at the one moment
|
|
1510
|
+
# it most wants a number
|
|
1511
|
+
detail = self.worker_manager.crash_details()
|
|
1512
|
+
logger.warning(f"Worker unavailable for memory status: {detail or e}")
|
|
1513
|
+
if detail is not None:
|
|
1514
|
+
# crash_details only answers for a process the OS has reaped,
|
|
1515
|
+
# so a worker that is merely slow to reply keeps its state -
|
|
1516
|
+
# a timeout is not evidence of death
|
|
1517
|
+
self.worker_manager.mark_crashed()
|
|
1518
|
+
return self._cached_memory("worker_unreachable")
|
|
1519
|
+
finally:
|
|
1520
|
+
self._worker_lock.release()
|
|
1521
|
+
if result.get("type") == "memory_status":
|
|
1522
|
+
self._record_memory(result.get("info"))
|
|
1523
|
+
return {
|
|
1524
|
+
"live": True,
|
|
1525
|
+
"info": self.last_memory,
|
|
1526
|
+
"stale": False,
|
|
1527
|
+
"reason": None,
|
|
1528
|
+
"age_seconds": 0.0,
|
|
1529
|
+
}
|
|
1530
|
+
return self._cached_memory("worker_unreachable")
|
|
1531
|
+
|
|
1532
|
+
def clear_memory(self, timeout=30):
|
|
1533
|
+
"""Drop every loaded pipeline and the step cache, then report the
|
|
1534
|
+
memory reading taken right after. Callers must check `is_busy()`
|
|
1535
|
+
first - this does not itself refuse a running/queued job, and racing
|
|
1536
|
+
one would clear state a queued run still expects resident. The
|
|
1537
|
+
30s timeout (vs. `memory_status`'s 5s) matches the REPL's `memory
|
|
1538
|
+
clear` (`repl_commands.py`): actually freeing CUDA memory takes
|
|
1539
|
+
longer than reading a counter does.
|
|
1540
|
+
|
|
1541
|
+
Returns the reading taken after the clear, or None when there was no
|
|
1542
|
+
worker to clear - nothing was resident in that case."""
|
|
1543
|
+
if not self.worker_manager.worker_active:
|
|
1544
|
+
# Nothing to clear, and not a fault: the pipelines and the step
|
|
1545
|
+
# cache both live in the worker process, so no worker running
|
|
1546
|
+
# means both are already gone. An on-demand worker is legitimately
|
|
1547
|
+
# absent on an idle server (#206), which this used to answer with
|
|
1548
|
+
# a 503 saying the worker was unavailable. None is "no reading
|
|
1549
|
+
# was taken", not a failure
|
|
1550
|
+
return None
|
|
1551
|
+
if not self._worker_lock.acquire(timeout=2):
|
|
1552
|
+
raise RuntimeError("worker busy")
|
|
1553
|
+
try:
|
|
1554
|
+
self.worker_manager.send_command({"type": "clear_memory"})
|
|
1555
|
+
result = self.worker_manager.get_result(timeout=timeout)
|
|
1556
|
+
finally:
|
|
1557
|
+
self._worker_lock.release()
|
|
1558
|
+
if result.get("type") != "memory_cleared":
|
|
1559
|
+
raise RuntimeError(f"unexpected worker reply: {result.get('type')}")
|
|
1560
|
+
self._record_memory(result.get("info"))
|
|
1561
|
+
return self.last_memory
|