diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/workflow.py
ADDED
|
@@ -0,0 +1,2007 @@
|
|
|
1
|
+
# Core functionality for loading and executing workflows
|
|
2
|
+
import os
|
|
3
|
+
import json
|
|
4
|
+
import torch
|
|
5
|
+
import copy
|
|
6
|
+
import gc
|
|
7
|
+
import hashlib
|
|
8
|
+
import logging
|
|
9
|
+
import secrets
|
|
10
|
+
from datetime import datetime, timezone
|
|
11
|
+
from .arguments import (
|
|
12
|
+
realize_args,
|
|
13
|
+
realize_constants,
|
|
14
|
+
fetch_constant,
|
|
15
|
+
is_constant_reference,
|
|
16
|
+
PREVIOUS_RESULT_PREFIX,
|
|
17
|
+
)
|
|
18
|
+
from .events import (
|
|
19
|
+
RunContext,
|
|
20
|
+
emit_phase,
|
|
21
|
+
WorkflowCancelled,
|
|
22
|
+
get_context,
|
|
23
|
+
current_context,
|
|
24
|
+
activate_context,
|
|
25
|
+
deactivate_context,
|
|
26
|
+
)
|
|
27
|
+
from .previous_results import (
|
|
28
|
+
StepResults,
|
|
29
|
+
previous_result_reference_errors,
|
|
30
|
+
)
|
|
31
|
+
from .locations import location_errors
|
|
32
|
+
from .reference_limits import reference_limit_errors
|
|
33
|
+
from .adapter_compatibility import adapter_errors, warn_adapters
|
|
34
|
+
from .elision import elide_definition, warn_elided
|
|
35
|
+
from .introspection import (
|
|
36
|
+
task_signature_errors,
|
|
37
|
+
component_type_errors,
|
|
38
|
+
component_name_errors,
|
|
39
|
+
)
|
|
40
|
+
from .dissolve_frame_errors import dissolve_frame_errors
|
|
41
|
+
from .task_domains import task_argument_errors
|
|
42
|
+
from .select_validation import select_errors
|
|
43
|
+
from .variable_constraints import (
|
|
44
|
+
apply_constraints,
|
|
45
|
+
constraint_errors,
|
|
46
|
+
constraint_reference_errors,
|
|
47
|
+
resolve_constraint_references,
|
|
48
|
+
)
|
|
49
|
+
from .result_fps import fps_errors
|
|
50
|
+
from .shots import step_shots
|
|
51
|
+
from .subfolders import step_subfolder, subfolder_errors
|
|
52
|
+
from .reference_names import reference_name_errors
|
|
53
|
+
from .video_extensions import video_extension_errors
|
|
54
|
+
from .content_types import content_type_errors
|
|
55
|
+
from .scalar_result_validation import scalar_result_errors
|
|
56
|
+
from .kernel_availability import kernel_availability_errors
|
|
57
|
+
from .vram_estimate import apply_vram_estimate, vram_estimate_errors
|
|
58
|
+
from .step import Step
|
|
59
|
+
from .step_cache import (
|
|
60
|
+
step_cache,
|
|
61
|
+
referenced_result_names,
|
|
62
|
+
reference_resolves_to,
|
|
63
|
+
normalized_downstream,
|
|
64
|
+
)
|
|
65
|
+
from .runs import (
|
|
66
|
+
FLAT_LAYOUT,
|
|
67
|
+
REALIZED_FILE_NAME,
|
|
68
|
+
activate_output_root,
|
|
69
|
+
assign_run_version,
|
|
70
|
+
deactivate_output_root,
|
|
71
|
+
workflow_identity,
|
|
72
|
+
manifest_relative_files,
|
|
73
|
+
new_run_id,
|
|
74
|
+
output_layout,
|
|
75
|
+
run_directory,
|
|
76
|
+
write_manifest,
|
|
77
|
+
write_realized_workflow,
|
|
78
|
+
)
|
|
79
|
+
from .realize import realize_workflow
|
|
80
|
+
from .schema import validate_data_all, format_validation_errors, load_schema
|
|
81
|
+
from .for_each import expand_for_each, ForEachError
|
|
82
|
+
from .variables import (
|
|
83
|
+
argument_errors,
|
|
84
|
+
replace_variables,
|
|
85
|
+
resolve_variable_values,
|
|
86
|
+
set_variables,
|
|
87
|
+
undeclared_variable_references,
|
|
88
|
+
VariableCycleError,
|
|
89
|
+
VariableNotFoundError,
|
|
90
|
+
)
|
|
91
|
+
from .pipeline_processors.pipeline import Pipeline
|
|
92
|
+
from .tasks.model_cache import clear_model_cache
|
|
93
|
+
from .tasks.task import Task
|
|
94
|
+
from . import get_device, empty_device_cache, device_memory_stats
|
|
95
|
+
from .host_memory import release_host_caches
|
|
96
|
+
from .security import (
|
|
97
|
+
validate_path,
|
|
98
|
+
validate_workflow_path,
|
|
99
|
+
validate_json_size,
|
|
100
|
+
validate_output_path,
|
|
101
|
+
SecurityError,
|
|
102
|
+
PathTraversalError,
|
|
103
|
+
InvalidInputError,
|
|
104
|
+
UntrustedWorkflowError,
|
|
105
|
+
)
|
|
106
|
+
from .workflow_sources import (
|
|
107
|
+
builtin_root,
|
|
108
|
+
catalog_root,
|
|
109
|
+
resolve_sub_workflow,
|
|
110
|
+
SubWorkflowNotFound,
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
logger = logging.getLogger("dw")
|
|
114
|
+
|
|
115
|
+
# The widest integer JavaScript's double represents exactly - the ceiling on
|
|
116
|
+
# any seed the engine draws, since seeds travel as JSON through a browser
|
|
117
|
+
SEED_BITS = 53
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class ConstantError(ValueError):
|
|
121
|
+
"""A 'constant:' variable default that failed to resolve during
|
|
122
|
+
validation, with the 'variables.<name>' path at fault."""
|
|
123
|
+
|
|
124
|
+
def __init__(self, path, message):
|
|
125
|
+
super().__init__(message)
|
|
126
|
+
self.path = path
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def workflow_from_file(file_spec, output_dir, workflow_dir=None):
|
|
130
|
+
"""Loads a workflow from a JSON file with security validation.
|
|
131
|
+
|
|
132
|
+
workflow_dir, when given, confines file_spec (and, via the returned
|
|
133
|
+
Workflow, any sub-workflow steps it references) to that directory - the
|
|
134
|
+
server passes its configured workflow_dir so a caller cannot escape it
|
|
135
|
+
via an inline workflow's base_dir or a sub-workflow step's path. CLI/REPL
|
|
136
|
+
callers leave it None: a locally-run workflow file is not a trust
|
|
137
|
+
boundary.
|
|
138
|
+
"""
|
|
139
|
+
logger.debug(f"Loading workflow from file: {file_spec}")
|
|
140
|
+
|
|
141
|
+
try:
|
|
142
|
+
# Validate file path and size
|
|
143
|
+
validated_path = validate_workflow_path(file_spec, workflow_dir)
|
|
144
|
+
validate_json_size(validated_path)
|
|
145
|
+
validated_output = validate_output_path(output_dir, None)
|
|
146
|
+
|
|
147
|
+
with open(validated_path, "r") as file:
|
|
148
|
+
workflow_data = json.load(file)
|
|
149
|
+
|
|
150
|
+
return Workflow(workflow_data, validated_output, validated_path, workflow_dir)
|
|
151
|
+
|
|
152
|
+
except SecurityError as e:
|
|
153
|
+
logger.error(f"Security validation failed for workflow {file_spec}: {e}")
|
|
154
|
+
raise
|
|
155
|
+
except (json.JSONDecodeError, OSError) as e:
|
|
156
|
+
logger.error(f"Failed to load workflow from {file_spec}: {e}")
|
|
157
|
+
raise
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def workflow_from_definition(
|
|
161
|
+
workflow_definition, output_dir, base_dir=None, workflow_dir=None
|
|
162
|
+
):
|
|
163
|
+
"""A Workflow from an inline definition (no file on disk).
|
|
164
|
+
|
|
165
|
+
The synthetic '__inline__.json' file_spec exists only to carry the
|
|
166
|
+
directory that relative paths inside the definition resolve against.
|
|
167
|
+
base_dir is caller-supplied (over HTTP, client-supplied) path-shaped
|
|
168
|
+
input, so it goes through the security validator like every other path -
|
|
169
|
+
confined to workflow_dir when the caller gives one, same as file_spec in
|
|
170
|
+
workflow_from_file, so a client cannot point an inline workflow's assets
|
|
171
|
+
(or a sub-workflow step it defines) anywhere on disk.
|
|
172
|
+
"""
|
|
173
|
+
validated_output = validate_output_path(output_dir, None)
|
|
174
|
+
if base_dir:
|
|
175
|
+
validated_base = validate_path(base_dir, workflow_dir, allow_create=False)
|
|
176
|
+
if not os.path.isdir(validated_base):
|
|
177
|
+
raise InvalidInputError(f"base_dir is not a directory: {base_dir}")
|
|
178
|
+
else:
|
|
179
|
+
# A confined run without a base_dir rests at the boundary itself, so
|
|
180
|
+
# the worker's re-validation of the stored base_dir agrees with this one
|
|
181
|
+
validated_base = os.path.abspath(workflow_dir) if workflow_dir else os.getcwd()
|
|
182
|
+
return Workflow(
|
|
183
|
+
workflow_definition,
|
|
184
|
+
validated_output,
|
|
185
|
+
os.path.join(validated_base, "__inline__.json"),
|
|
186
|
+
workflow_dir,
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def workflow_output_subfolder(file_spec):
|
|
191
|
+
"""The subfolder a workflow's outputs land in, mirroring its position
|
|
192
|
+
under the nearest directory literally named 'workflows' in its path.
|
|
193
|
+
|
|
194
|
+
'workflows/ltx/Foo.json' -> 'ltx'; 'workflows/Foo.json' (or a builtin,
|
|
195
|
+
always dw/workflows/<name>.json) -> '' (flat, no spurious subfolder);
|
|
196
|
+
a path with no 'workflows' segment at all (an inline definition's
|
|
197
|
+
synthetic file_spec, say) -> '' as a fallback. The *last* 'workflows'
|
|
198
|
+
segment wins, matching the packaged dw/workflows tree when a checkout
|
|
199
|
+
also has a top-level workflows/ directory somewhere in its ancestry.
|
|
200
|
+
"""
|
|
201
|
+
if not file_spec:
|
|
202
|
+
return ""
|
|
203
|
+
|
|
204
|
+
directory = os.path.dirname(os.path.abspath(file_spec))
|
|
205
|
+
parts = os.path.normpath(directory).split(os.sep)
|
|
206
|
+
try:
|
|
207
|
+
index = len(parts) - 1 - parts[::-1].index("workflows")
|
|
208
|
+
except ValueError:
|
|
209
|
+
return ""
|
|
210
|
+
|
|
211
|
+
return os.path.join(*parts[index + 1 :]) if index + 1 < len(parts) else ""
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def catalog_root_dir(file_spec):
|
|
215
|
+
"""The nearest ancestor directory literally named 'workflows' of
|
|
216
|
+
file_spec, else file_spec's own directory.
|
|
217
|
+
|
|
218
|
+
Used to confine a relative sub-workflow reference when a run carries no
|
|
219
|
+
workflow_dir of its own (an unconfined CLI run) - the same "last
|
|
220
|
+
'workflows' segment" rule workflow_output_subfolder uses for output
|
|
221
|
+
naming, but returning the directory itself rather than what sits under
|
|
222
|
+
it. It is `catalog_root` asked for a file rather than a directory, so
|
|
223
|
+
the resolver (dw/workflow_sources.py) confines to exactly this root.
|
|
224
|
+
"""
|
|
225
|
+
return catalog_root(os.path.dirname(os.path.abspath(file_spec)))
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def pipeline_cache_key(pipeline_definition):
|
|
229
|
+
"""Stable identity for a loaded pipeline.
|
|
230
|
+
|
|
231
|
+
Hashes everything that shapes loading - configuration, components,
|
|
232
|
+
quantization, loras - and excludes what varies per call (arguments, seed,
|
|
233
|
+
chain), so a cache hit means "this exact model stack is already loaded".
|
|
234
|
+
Keying the cache by identity instead of step name means two workflows
|
|
235
|
+
whose steps happen to share a name can no longer collide, and a rerun of
|
|
236
|
+
an edited workflow keeps every pipeline whose definition did not change.
|
|
237
|
+
|
|
238
|
+
Computed after variable substitution but the excluded keys keep realized
|
|
239
|
+
per-run values (images, generators) out of the hash; realized types and
|
|
240
|
+
dtypes stringify stably via default=str.
|
|
241
|
+
"""
|
|
242
|
+
load_definition = {
|
|
243
|
+
k: v
|
|
244
|
+
for k, v in pipeline_definition.items()
|
|
245
|
+
if k not in ("arguments", "seed", "chain")
|
|
246
|
+
}
|
|
247
|
+
serialized = json.dumps(load_definition, sort_keys=True, default=str)
|
|
248
|
+
return hashlib.sha256(serialized.encode()).hexdigest()
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _allocated_mb():
|
|
252
|
+
"""Device memory in use right now, for the pipeline_released event -
|
|
253
|
+
None where the backend cannot say, so a reading is never confused with
|
|
254
|
+
a genuine zero."""
|
|
255
|
+
try:
|
|
256
|
+
stats = device_memory_stats()
|
|
257
|
+
except Exception: # a progress figure is never worth failing a run over
|
|
258
|
+
return None
|
|
259
|
+
return stats["allocated_mb"] if stats["available"] else None
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _release_host_caches(step_name):
|
|
263
|
+
"""Hand the host memory a release freed back to the OS, not at job end.
|
|
264
|
+
|
|
265
|
+
`release_host_caches` only touches blocks nothing is using, so anything
|
|
266
|
+
still loaded is undisturbed. A cleanup is never worth failing a run for.
|
|
267
|
+
"""
|
|
268
|
+
try:
|
|
269
|
+
released = release_host_caches()
|
|
270
|
+
except Exception as e:
|
|
271
|
+
logger.debug(f"Could not release host caches after {step_name}: {e}")
|
|
272
|
+
return
|
|
273
|
+
if released:
|
|
274
|
+
logger.info(f"Release after {step_name} returned {released:.0f} MB to the OS")
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _relative_shots(entry, run_dir):
|
|
278
|
+
"""A manifest entry's `shots`, each `file` made relative as `files` is."""
|
|
279
|
+
shots = entry.get("shots")
|
|
280
|
+
if not shots or not any("file" in shot for shot in shots):
|
|
281
|
+
return {}
|
|
282
|
+
files = manifest_relative_files([shot["file"] for shot in shots], run_dir)
|
|
283
|
+
return {"shots": [{**shot, "file": f} for shot, f in zip(shots, files)]}
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def selected_field(step_data, selected):
|
|
287
|
+
"""The manifest/step_end 'selected' block for a step's Result.selected.
|
|
288
|
+
|
|
289
|
+
Carries the winning position and score always; adds 'entry' - the
|
|
290
|
+
for_each member name ('shot@b') - only when the step's 'candidates'
|
|
291
|
+
argument was built from a gather: reference, recoverable at this point
|
|
292
|
+
only because step_data still holds the post-for_each-expansion,
|
|
293
|
+
pre-argument-resolution reference strings (dw/for_each.py's _gather()).
|
|
294
|
+
"""
|
|
295
|
+
if selected is None:
|
|
296
|
+
return None
|
|
297
|
+
|
|
298
|
+
field = dict(selected)
|
|
299
|
+
candidates = step_data.get("task", {}).get("arguments", {}).get("candidates")
|
|
300
|
+
if isinstance(candidates, list):
|
|
301
|
+
position = selected.get("position")
|
|
302
|
+
if isinstance(position, int) and 0 <= position < len(candidates):
|
|
303
|
+
candidate = candidates[position]
|
|
304
|
+
if isinstance(candidate, str) and candidate.startswith(
|
|
305
|
+
PREVIOUS_RESULT_PREFIX
|
|
306
|
+
):
|
|
307
|
+
field["entry"] = candidate[len(PREVIOUS_RESULT_PREFIX) :]
|
|
308
|
+
return field
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def release_unreferenced_results(results, remaining_refs):
|
|
312
|
+
"""Drop results no remaining reference can resolve to.
|
|
313
|
+
|
|
314
|
+
A reference resolves to a result whose name it equals or extends with a
|
|
315
|
+
property ('step.mask'), so any result that is such a prefix stays. Saved
|
|
316
|
+
artifacts are already on disk - holding every intermediate image and frame
|
|
317
|
+
list in RAM until the workflow ends is what OOMs long chains.
|
|
318
|
+
"""
|
|
319
|
+
for name in [
|
|
320
|
+
n
|
|
321
|
+
for n in results
|
|
322
|
+
if not any(reference_resolves_to(ref, n) for ref in remaining_refs)
|
|
323
|
+
]:
|
|
324
|
+
logger.debug(f"Releasing result: {name}")
|
|
325
|
+
del results[name]
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
class Workflow:
|
|
329
|
+
"""
|
|
330
|
+
Main class for managing and executing workflows defined in JSON format
|
|
331
|
+
Handles variable substitution, step execution, and result management
|
|
332
|
+
"""
|
|
333
|
+
|
|
334
|
+
# Whether the workflow that delegated to this one had the step cache
|
|
335
|
+
# enabled. A top-level run has no parent and decides for itself; a
|
|
336
|
+
# sub-workflow's parent overwrites this in create_step_action
|
|
337
|
+
_cache_enabled_by_parent = True
|
|
338
|
+
# What run() decided for the run in progress, read by create_step_action
|
|
339
|
+
# to hand down to a sub-workflow
|
|
340
|
+
_cache_enabled_this_run = True
|
|
341
|
+
|
|
342
|
+
# The directory the run in progress writes into, set by run() and, for a
|
|
343
|
+
# sub-workflow, handed down by the parent - one execution is one
|
|
344
|
+
# directory, whichever workflow inside it did the writing. None in the
|
|
345
|
+
# flat layout, and before a run starts
|
|
346
|
+
_run_dir = None
|
|
347
|
+
# Whether that directory came from a parent workflow. A sub-workflow is
|
|
348
|
+
# part of the parent's execution: it writes into the same directory and
|
|
349
|
+
# leaves no manifest of its own, since its steps are already rolled up
|
|
350
|
+
# into the parent's
|
|
351
|
+
_run_dir_inherited = False
|
|
352
|
+
# That directory's ordinal among this workflow's runs - what the gallery
|
|
353
|
+
# shows as 'v4'. None in the flat layout, for a sub-workflow (which is
|
|
354
|
+
# part of the parent's run, not a run of its own), and before a run
|
|
355
|
+
# starts
|
|
356
|
+
_run_version = None
|
|
357
|
+
# Where the parent step that delegated to this workflow sits in the
|
|
358
|
+
# run the caller queued: {"step", "index", "total_steps"}. A child
|
|
359
|
+
# counts its own steps from zero, so without this a composed run
|
|
360
|
+
# reported "step 1 of 1" from inside the first of the parent's three
|
|
361
|
+
# (#90) - and "is this nearly finished" is the whole question progress
|
|
362
|
+
# answers. Handed straight down to a grandchild, so the numbers always
|
|
363
|
+
# describe the run that was queued
|
|
364
|
+
_parent_progress = None
|
|
365
|
+
# Whether the parent step that composed this workflow declares a
|
|
366
|
+
# `result` of its own. It does the saving then, and this run's last step
|
|
367
|
+
# does not: the two used to write the same artifact twice, once under
|
|
368
|
+
# the parent step's name and subfolder and once under the child's, with
|
|
369
|
+
# the child's entry shadowing a manifest key the caller never wrote
|
|
370
|
+
# (#92). A child whose parent declares nothing still saves, since
|
|
371
|
+
# otherwise the output would exist nowhere
|
|
372
|
+
_final_save_owned_by_parent = False
|
|
373
|
+
|
|
374
|
+
def __init__(self, workflow_definition, output_dir, file_spec, workflow_dir=None):
|
|
375
|
+
self.workflow_definition = workflow_definition
|
|
376
|
+
self.output_dir = output_dir
|
|
377
|
+
self.file_spec = file_spec
|
|
378
|
+
# Confines sub-workflow step resolution (below) when set - the
|
|
379
|
+
# server passes its configured workflow_dir; CLI/REPL callers leave
|
|
380
|
+
# it None since a locally-run workflow is not a trust boundary
|
|
381
|
+
self.workflow_dir = workflow_dir
|
|
382
|
+
|
|
383
|
+
@property
|
|
384
|
+
def name(self):
|
|
385
|
+
return self.workflow_definition.get("id", "unknown")
|
|
386
|
+
|
|
387
|
+
@property
|
|
388
|
+
def argument_template(self):
|
|
389
|
+
return self.workflow_definition.get("argument_template", {})
|
|
390
|
+
|
|
391
|
+
@property
|
|
392
|
+
def variables(self):
|
|
393
|
+
return self.workflow_definition.get("variables", {})
|
|
394
|
+
|
|
395
|
+
def step_save_name(self, workflow_id, step_name, index):
|
|
396
|
+
"""The base name a step's files are written under.
|
|
397
|
+
|
|
398
|
+
Inside a composed child the parent step's name leads, so two steps
|
|
399
|
+
composing the same workflow do not both want one name and get told
|
|
400
|
+
apart by a '-2' suffix that says nothing about which step made it
|
|
401
|
+
(#92).
|
|
402
|
+
"""
|
|
403
|
+
base = f"{workflow_id}-{step_name}.{index}"
|
|
404
|
+
parent = self._parent_progress
|
|
405
|
+
return f"{parent['step']}.{base}" if parent else base
|
|
406
|
+
|
|
407
|
+
def _parent_progress_fields(self):
|
|
408
|
+
"""The queued run's own step counter, on an event a sub-workflow
|
|
409
|
+
emits - empty for a top-level run, whose index is already that."""
|
|
410
|
+
parent = self._parent_progress
|
|
411
|
+
if not parent:
|
|
412
|
+
return {}
|
|
413
|
+
return {
|
|
414
|
+
"parent_step": parent["step"],
|
|
415
|
+
"parent_index": parent["index"],
|
|
416
|
+
"parent_total_steps": parent["total_steps"],
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
def step_file_prefix(self, step_name):
|
|
420
|
+
"""Naming prefix for files a step writes on its own (chain segment
|
|
421
|
+
spills), matching the workflow-id-step naming its results are saved
|
|
422
|
+
under."""
|
|
423
|
+
prefix = f"{self.name}-{step_name}"
|
|
424
|
+
parent = self._parent_progress
|
|
425
|
+
return f"{parent['step']}.{prefix}" if parent else prefix
|
|
426
|
+
|
|
427
|
+
@property
|
|
428
|
+
def effective_output_dir(self):
|
|
429
|
+
"""Where this workflow's own results are written.
|
|
430
|
+
|
|
431
|
+
In the default layout that is the run directory run() opened -
|
|
432
|
+
'<output_dir>/<identity>/<run id>/' - shared by every step of the
|
|
433
|
+
run, sub-workflows included, so one execution leaves one directory.
|
|
434
|
+
|
|
435
|
+
In the flat layout it is output_dir plus a subfolder mirroring the
|
|
436
|
+
workflow file's position under a 'workflows' directory, if it has
|
|
437
|
+
one: a workflow at 'workflows/ltx/Foo.json' writes under
|
|
438
|
+
'<output_dir>/ltx/'; one directly inside a 'workflows' folder (or a
|
|
439
|
+
builtin, which always resolves to dw/workflows/<name>.json) writes
|
|
440
|
+
flat at '<output_dir>/', same as one outside any 'workflows' tree
|
|
441
|
+
entirely. self.output_dir itself always stays the plain root - this
|
|
442
|
+
is derived fresh from it every time, so a flat-layout sub-workflow
|
|
443
|
+
computes its own subfolder from its own file, not the parent's.
|
|
444
|
+
"""
|
|
445
|
+
if self._run_dir:
|
|
446
|
+
return self._run_dir
|
|
447
|
+
subfolder = workflow_output_subfolder(self.file_spec)
|
|
448
|
+
return (
|
|
449
|
+
os.path.join(self.output_dir, subfolder) if subfolder else self.output_dir
|
|
450
|
+
)
|
|
451
|
+
|
|
452
|
+
def step_output_dir(self, step_definition):
|
|
453
|
+
"""Where one step writes: the run directory, or the subfolder of it
|
|
454
|
+
the step's result names.
|
|
455
|
+
|
|
456
|
+
Computed here, once, rather than inside Result.save, because two
|
|
457
|
+
things write on a step's behalf - Result.save for its results and
|
|
458
|
+
the pipeline wrapper for a chain's save_segments spill - and both
|
|
459
|
+
have to land in the same place. The shape was checked statically by
|
|
460
|
+
validation_errors; it is checked again here for a definition that
|
|
461
|
+
reached the engine without it, and containment (that the joined
|
|
462
|
+
path is really inside the run directory) is checked on the join.
|
|
463
|
+
"""
|
|
464
|
+
base = self.effective_output_dir
|
|
465
|
+
subfolder = step_subfolder(step_definition)
|
|
466
|
+
if not subfolder:
|
|
467
|
+
return base
|
|
468
|
+
target = validate_output_path(os.path.join(base, subfolder), base)
|
|
469
|
+
os.makedirs(target, exist_ok=True)
|
|
470
|
+
return target
|
|
471
|
+
|
|
472
|
+
def expanded_definition(self, arguments=None, source_indices=None):
|
|
473
|
+
"""The definition as the run will see it: constants realized,
|
|
474
|
+
variables substituted - the caller's `arguments` folded in when they
|
|
475
|
+
are all good, else the declared defaults - and every for_each step
|
|
476
|
+
expanded.
|
|
477
|
+
|
|
478
|
+
Raises ForEachError for a for_each that cannot be expanded,
|
|
479
|
+
ConstantError for a 'constant:' variable default that fails to
|
|
480
|
+
resolve, and VariableNotFoundError for a 'variable:' that names
|
|
481
|
+
nothing - which is exactly what the run itself would raise, since a
|
|
482
|
+
definition that declares variables is always substituted before it
|
|
483
|
+
runs.
|
|
484
|
+
|
|
485
|
+
`source_indices`, when a list is passed, comes back holding the
|
|
486
|
+
index in *this* definition's steps of every expanded step, so an
|
|
487
|
+
error can be reported at a path in the file the author wrote.
|
|
488
|
+
"""
|
|
489
|
+
definition = copy.deepcopy(self.workflow_definition)
|
|
490
|
+
variables = definition.get("variables")
|
|
491
|
+
if isinstance(variables, dict):
|
|
492
|
+
# the run realizes constants before folding arguments, and a
|
|
493
|
+
# list defaulted to a 'constant:' name must expand here as it
|
|
494
|
+
# does there - a name lookup, no download. Realizing a constant
|
|
495
|
+
# imports the module it names, so validating one runs the same
|
|
496
|
+
# trust gate (require_trusted_dotted_name) a run would - only
|
|
497
|
+
# the diffusers ecosystem allowlist, unless the caller trusts
|
|
498
|
+
# the workflow. Realized per top-level variable, not as one
|
|
499
|
+
# call over the whole dict, so a failure names the variable.
|
|
500
|
+
for name, value in variables.items():
|
|
501
|
+
try:
|
|
502
|
+
if is_constant_reference(value):
|
|
503
|
+
variables[name] = fetch_constant(value)
|
|
504
|
+
else:
|
|
505
|
+
realize_constants(value)
|
|
506
|
+
except (ValueError, InvalidInputError, UntrustedWorkflowError) as e:
|
|
507
|
+
raise ConstantError(f"variables.{name}", str(e)) from e
|
|
508
|
+
if arguments and not argument_errors(definition, arguments):
|
|
509
|
+
set_variables(arguments, variables)
|
|
510
|
+
variables = resolve_variable_values(variables)
|
|
511
|
+
definition = replace_variables(definition, variables)
|
|
512
|
+
return expand_for_each(definition, source_indices)
|
|
513
|
+
|
|
514
|
+
def resolve_sub_workflow_path(self, path):
|
|
515
|
+
"""Where one sub-workflow step's `path` resolves to, as
|
|
516
|
+
(path, root) - the same resolution create_step_action does, asked
|
|
517
|
+
ahead of the run so validation can answer for free what used to cost
|
|
518
|
+
a queued job to find out (#89).
|
|
519
|
+
|
|
520
|
+
Raises SubWorkflowNotFound, SecurityError or InvalidInputError,
|
|
521
|
+
each carrying the message the run would have failed with.
|
|
522
|
+
"""
|
|
523
|
+
confine_to = self.workflow_dir
|
|
524
|
+
if path.startswith("builtin:"):
|
|
525
|
+
builtin_name = path.replace("builtin:", "")
|
|
526
|
+
if (
|
|
527
|
+
not builtin_name.endswith(".json")
|
|
528
|
+
or "/" in builtin_name
|
|
529
|
+
or "\\" in builtin_name
|
|
530
|
+
):
|
|
531
|
+
raise InvalidInputError(
|
|
532
|
+
f"Invalid builtin workflow name: {builtin_name}. It must "
|
|
533
|
+
"be a bare '<name>.json' filename with no path segments - "
|
|
534
|
+
f"'builtin:' only looks in the packaged workflows root: "
|
|
535
|
+
f"{builtin_root()}"
|
|
536
|
+
)
|
|
537
|
+
confine_to = builtin_root()
|
|
538
|
+
resolved = os.path.join(confine_to, builtin_name)
|
|
539
|
+
if not os.path.isfile(resolved):
|
|
540
|
+
raise SubWorkflowNotFound(path, [resolved])
|
|
541
|
+
return validate_workflow_path(resolved, confine_to), confine_to
|
|
542
|
+
if confine_to is None and not os.path.isabs(path):
|
|
543
|
+
confine_to = catalog_root_dir(self.file_spec)
|
|
544
|
+
resolved, confine_to = resolve_sub_workflow(
|
|
545
|
+
path, os.path.dirname(self.file_spec), confine_to
|
|
546
|
+
)
|
|
547
|
+
return validate_workflow_path(resolved, confine_to), confine_to
|
|
548
|
+
|
|
549
|
+
def sub_workflow_errors(self, expanded, source_indices=None, composing=None):
|
|
550
|
+
"""Every sub-workflow step whose `path` names nothing this server can
|
|
551
|
+
reach, composes a workflow already on the chain, or resolves to a
|
|
552
|
+
workflow that does not itself validate.
|
|
553
|
+
|
|
554
|
+
`composing` is the resolved path of every workflow above this one,
|
|
555
|
+
which is what makes a cycle an error here rather than a recursion
|
|
556
|
+
the run discovers.
|
|
557
|
+
"""
|
|
558
|
+
errors = []
|
|
559
|
+
composing = list(composing or [])
|
|
560
|
+
for index, step in enumerate(expanded.get("steps", []) or []):
|
|
561
|
+
reference = step.get("workflow")
|
|
562
|
+
if not isinstance(reference, dict) or not isinstance(
|
|
563
|
+
reference.get("path"), str
|
|
564
|
+
):
|
|
565
|
+
continue
|
|
566
|
+
source = source_indices[index] if source_indices else index
|
|
567
|
+
where = f"steps[{source}].workflow.path"
|
|
568
|
+
path = reference["path"]
|
|
569
|
+
try:
|
|
570
|
+
resolved, root = self.resolve_sub_workflow_path(path)
|
|
571
|
+
except (SubWorkflowNotFound, SecurityError, InvalidInputError) as e:
|
|
572
|
+
errors.append({"path": where, "message": str(e)})
|
|
573
|
+
continue
|
|
574
|
+
if resolved in composing:
|
|
575
|
+
errors.append(
|
|
576
|
+
{
|
|
577
|
+
"path": where,
|
|
578
|
+
"message": (
|
|
579
|
+
f"Sub-workflow '{path}' composes a workflow that "
|
|
580
|
+
"is already composing it - a cycle: "
|
|
581
|
+
+ " -> ".join(composing + [resolved])
|
|
582
|
+
),
|
|
583
|
+
}
|
|
584
|
+
)
|
|
585
|
+
continue
|
|
586
|
+
try:
|
|
587
|
+
child = workflow_from_file(resolved, self.output_dir, root)
|
|
588
|
+
except Exception as e:
|
|
589
|
+
errors.append({"path": where, "message": f"Sub-workflow '{path}': {e}"})
|
|
590
|
+
continue
|
|
591
|
+
for error in child.validation_errors(composing=composing + [resolved]):
|
|
592
|
+
errors.append(
|
|
593
|
+
{
|
|
594
|
+
"path": f"{where} -> {error['path']}",
|
|
595
|
+
"message": f"Sub-workflow '{path}': {error['message']}",
|
|
596
|
+
}
|
|
597
|
+
)
|
|
598
|
+
return errors
|
|
599
|
+
|
|
600
|
+
def sub_workflow_warnings(self, expanded=None):
|
|
601
|
+
"""An argument a sub-workflow step passes down that the workflow it
|
|
602
|
+
composes declares no variable for - dropped in silence at run time,
|
|
603
|
+
and composition is exactly where a name drifts (#89)."""
|
|
604
|
+
warnings = []
|
|
605
|
+
try:
|
|
606
|
+
expanded = expanded if expanded is not None else self.expanded_definition()
|
|
607
|
+
except Exception:
|
|
608
|
+
return warnings
|
|
609
|
+
for index, step in enumerate(expanded.get("steps", []) or []):
|
|
610
|
+
reference = step.get("workflow")
|
|
611
|
+
if not isinstance(reference, dict):
|
|
612
|
+
continue
|
|
613
|
+
passed = reference.get("arguments")
|
|
614
|
+
if not isinstance(passed, dict) or not isinstance(
|
|
615
|
+
reference.get("path"), str
|
|
616
|
+
):
|
|
617
|
+
continue
|
|
618
|
+
try:
|
|
619
|
+
resolved, root = self.resolve_sub_workflow_path(reference["path"])
|
|
620
|
+
child = workflow_from_file(resolved, self.output_dir, root)
|
|
621
|
+
except Exception:
|
|
622
|
+
# An unresolvable path is an error, reported by
|
|
623
|
+
# sub_workflow_errors - not a second complaint here
|
|
624
|
+
continue
|
|
625
|
+
declared = child.workflow_definition.get("variables") or {}
|
|
626
|
+
for name in sorted(set(passed) - set(declared)):
|
|
627
|
+
warnings.append(
|
|
628
|
+
{
|
|
629
|
+
"path": f"steps[{index}].workflow.arguments.{name}",
|
|
630
|
+
"message": (
|
|
631
|
+
f"'{reference['path']}' declares no variable "
|
|
632
|
+
f"'{name}' - the value is dropped. Declared: "
|
|
633
|
+
+ (", ".join(sorted(declared)) or "<none>")
|
|
634
|
+
),
|
|
635
|
+
}
|
|
636
|
+
)
|
|
637
|
+
return warnings
|
|
638
|
+
|
|
639
|
+
def validation_errors(self, arguments=None, composing=None):
|
|
640
|
+
"""Every schema violation in the definition, as [{path, message}];
|
|
641
|
+
empty when it validates. `arguments` are the caller's, so a
|
|
642
|
+
for_each over a list the caller supplies is checked as it will run.
|
|
643
|
+
|
|
644
|
+
`composing` carries the chain of sub-workflows above this one, so a
|
|
645
|
+
workflow that composes itself is an error rather than a recursion.
|
|
646
|
+
"""
|
|
647
|
+
errors = validate_data_all(self.workflow_definition, load_schema("workflow"))
|
|
648
|
+
# Only once the shape is known good: the passes below walk the
|
|
649
|
+
# steps array and a definition that fails the schema may have no
|
|
650
|
+
# such array to walk
|
|
651
|
+
if errors:
|
|
652
|
+
return errors
|
|
653
|
+
source_indices = []
|
|
654
|
+
try:
|
|
655
|
+
expanded = self.expanded_definition(arguments, source_indices)
|
|
656
|
+
except ForEachError as e:
|
|
657
|
+
return [{"path": e.path, "message": str(e)}]
|
|
658
|
+
except ConstantError as e:
|
|
659
|
+
return [{"path": e.path, "message": str(e)}]
|
|
660
|
+
except VariableNotFoundError:
|
|
661
|
+
# Every undeclared reference, not just the first one substitution
|
|
662
|
+
# tripped over - and reported where each sits rather than as a
|
|
663
|
+
# for_each whose list arrived unsubstituted, which is what a
|
|
664
|
+
# half-substituted definition used to look like from here
|
|
665
|
+
return self._undeclared_variable_errors(arguments)
|
|
666
|
+
except VariableCycleError as e:
|
|
667
|
+
# resolve_variable_values raises this for a variable that
|
|
668
|
+
# references itself, directly or through others - there is
|
|
669
|
+
# no single path inside the definition to blame, so it is
|
|
670
|
+
# reported against 'variables' as a whole rather than escaping
|
|
671
|
+
# as an unhandled exception
|
|
672
|
+
return [{"path": "variables", "message": str(e)}]
|
|
673
|
+
base_dir = (
|
|
674
|
+
os.path.dirname(os.path.abspath(self.file_spec)) if self.file_spec else None
|
|
675
|
+
)
|
|
676
|
+
task_errors = task_signature_errors(
|
|
677
|
+
expanded, source_indices, self.workflow_definition
|
|
678
|
+
)
|
|
679
|
+
if arguments is None:
|
|
680
|
+
# A step that feeds a required argument from `variable:name` and
|
|
681
|
+
# a variable whose default is null is a fine document - the
|
|
682
|
+
# variable just hasn't been given a value yet, which is exactly
|
|
683
|
+
# what no-arguments means here (save_workflow, or
|
|
684
|
+
# validate_workflow called to check the document rather than a
|
|
685
|
+
# specific run). Downgraded to a warning
|
|
686
|
+
# (null_variable_argument_warnings) rather than dropped outright,
|
|
687
|
+
# since it is still true a run left as-is would fail (#364).
|
|
688
|
+
# Anything else task_signature_errors reports - a genuinely
|
|
689
|
+
# missing or unknown argument - stays a hard error regardless
|
|
690
|
+
task_errors = [e for e in task_errors if "variable" not in e]
|
|
691
|
+
return (
|
|
692
|
+
previous_result_reference_errors(expanded, source_indices)
|
|
693
|
+
+ subfolder_errors(expanded, source_indices)
|
|
694
|
+
+ fps_errors(expanded, source_indices)
|
|
695
|
+
# A reference name no workspace could ever resolve - the '@' a
|
|
696
|
+
# for_each member's own file carries, rejected after the queue
|
|
697
|
+
# by a message that named a valid form and not the objection
|
|
698
|
+
# (dw/reference_names.py, #162)
|
|
699
|
+
+ reference_name_errors(expanded, source_indices)
|
|
700
|
+
# A still image handed to a 'video' argument by path or asset:/
|
|
701
|
+
# output: reference validated clean and then died inside
|
|
702
|
+
# fetch_video's extension gate in the first seconds of the run -
|
|
703
|
+
# refused here for the cases the extension is already knowable
|
|
704
|
+
# (dw/video_extensions.py, #347)
|
|
705
|
+
+ video_extension_errors(expanded, source_indices)
|
|
706
|
+
# A result content_type no writer will accept - a bare word like
|
|
707
|
+
# "video" validated clean and then died inside the writer with a
|
|
708
|
+
# traceback naming neither the field nor the value
|
|
709
|
+
# (dw/content_types.py, #168)
|
|
710
|
+
+ content_type_errors(expanded, source_indices)
|
|
711
|
+
# A 'result' block on a step whose command returns a scalar, not
|
|
712
|
+
# an artifact - judge's score validated clean and then died
|
|
713
|
+
# inside save_artifact with a bare TypeError after the fan-out
|
|
714
|
+
# ahead of it had already generated (dw/scalar_result_validation.py,
|
|
715
|
+
# #212)
|
|
716
|
+
+ scalar_result_errors(expanded, source_indices)
|
|
717
|
+
# A location policy refuses before a model load is spent on the
|
|
718
|
+
# run rather than after it (dw/locations.py)
|
|
719
|
+
+ location_errors(expanded, source_indices, base_dir)
|
|
720
|
+
# A reference set the pipeline would refuse costs a checkpoint
|
|
721
|
+
# load to find out about otherwise (dw/reference_limits.py, #136)
|
|
722
|
+
+ reference_limit_errors(expanded, source_indices)
|
|
723
|
+
# An adapter trained for the other checkpoint partition, which
|
|
724
|
+
# the pipeline loads without complaint and answers worse for -
|
|
725
|
+
# the one H3 mistake that never shows in the output
|
|
726
|
+
# (dw/adapter_compatibility.py, #155)
|
|
727
|
+
+ adapter_errors(
|
|
728
|
+
expanded,
|
|
729
|
+
source_indices,
|
|
730
|
+
written=self.workflow_definition,
|
|
731
|
+
supplied=set(arguments or {}),
|
|
732
|
+
)
|
|
733
|
+
# A number outside a task argument's declared domain is refused
|
|
734
|
+
# here rather than interpreted at run time - a negative frame
|
|
735
|
+
# count was a Python slice from the end of the track and a zero
|
|
736
|
+
# sample rate a silent fallback to 44100 (dw/task_domains.py,
|
|
737
|
+
# #139, #140)
|
|
738
|
+
+ task_argument_errors(expanded, source_indices)
|
|
739
|
+
# A dissolve_videos overlap wider than a statically-resolvable
|
|
740
|
+
# input's real frame count decoded clean past the queue and
|
|
741
|
+
# failed only after every upstream step had already generated -
|
|
742
|
+
# refused here for a literal dissolve_frames against an asset:/
|
|
743
|
+
# output:/literal-path video, the cases the frame count is
|
|
744
|
+
# already knowable (dw/dissolve_frame_errors.py, #400)
|
|
745
|
+
+ dissolve_frame_errors(expanded, source_indices, base_dir)
|
|
746
|
+
# A select step whose rule is misspelled, or whose
|
|
747
|
+
# threshold/index does not match its rule, validated clean and
|
|
748
|
+
# died on select's own run-time ValueError after the fan-out
|
|
749
|
+
# ahead of it had already generated (dw/select_validation.py,
|
|
750
|
+
# docs/proposals/score-and-select.md)
|
|
751
|
+
+ select_errors(expanded, source_indices)
|
|
752
|
+
# A required task argument left unset validated as `valid: true`
|
|
753
|
+
# and then failed the job on Python's own signature error, which
|
|
754
|
+
# is the one mistake a free pre-flight most obviously exists for
|
|
755
|
+
# (dw/introspection.py, #141). When the step supplies it by
|
|
756
|
+
# `variable:name` and only the variable's value is null, the
|
|
757
|
+
# error carries a `variable` key (#364) so a caller checking the
|
|
758
|
+
# document itself - no arguments of its own - can tell "the
|
|
759
|
+
# variable needs a value at run time" apart from "the step is
|
|
760
|
+
# broken", and downgrade the former below
|
|
761
|
+
+ task_errors
|
|
762
|
+
# A step's pipeline names a component_type/scheduler_type/
|
|
763
|
+
# config_type that does not exist (or is outside the trusted
|
|
764
|
+
# ecosystem entirely) - validated clean and died 3s into the run
|
|
765
|
+
# after a checkpoint the plan had already quoted for downloading
|
|
766
|
+
# (dw/introspection.py, #345)
|
|
767
|
+
+ component_type_errors(expanded, source_indices)
|
|
768
|
+
# A step's pipeline configures a component (`configuration.
|
|
769
|
+
# components`) its component_type does not register - validated
|
|
770
|
+
# clean and died 3s into the run's `loading` phase, after a
|
|
771
|
+
# checkpoint (and for an IC-LoRA step, LoRA weights) the plan had
|
|
772
|
+
# already quoted for downloading (dw/introspection.py, #442)
|
|
773
|
+
+ component_name_errors(expanded, source_indices)
|
|
774
|
+
# A value outside a rule the workflow declares - the bound that
|
|
775
|
+
# cost 138 s of loading to discover, refused for free at the
|
|
776
|
+
# path the value sits at (dw/variable_constraints.py, #96)
|
|
777
|
+
+ constraint_errors(
|
|
778
|
+
self.workflow_definition, arguments, supplied=set(arguments or {})
|
|
779
|
+
)
|
|
780
|
+
+ constraint_reference_errors(self.workflow_definition)
|
|
781
|
+
# A (width, height, num_frames)-shaped combination a declared
|
|
782
|
+
# vram_estimate projects past the card 'cost' was measured on -
|
|
783
|
+
# refused here rather than found 90+ seconds into denoising on
|
|
784
|
+
# an OOM the caller had no way to see coming (dw/vram_estimate.py,
|
|
785
|
+
# #265)
|
|
786
|
+
+ vram_estimate_errors(
|
|
787
|
+
self.workflow_definition, arguments, supplied=set(arguments or {})
|
|
788
|
+
)
|
|
789
|
+
# An 'attn_processor_type' whose Hub kernel this machine has no
|
|
790
|
+
# build variant for - validated clean and then died 88s into
|
|
791
|
+
# loading, naming a torch/natten mismatch the construction alone
|
|
792
|
+
# would have said in under two seconds (dw/kernel_availability.py,
|
|
793
|
+
# #178)
|
|
794
|
+
+ kernel_availability_errors(expanded, source_indices)
|
|
795
|
+
+ self.sub_workflow_errors(expanded, source_indices, composing)
|
|
796
|
+
)
|
|
797
|
+
|
|
798
|
+
def adapter_warnings(self, arguments=None):
|
|
799
|
+
"""Every adapter whose file name says nothing about which checkpoint
|
|
800
|
+
partition it was trained for - valid, and worth saying, since
|
|
801
|
+
nothing at run time will (#155).
|
|
802
|
+
|
|
803
|
+
Best effort: a definition the schema or the expander refuses has its
|
|
804
|
+
own errors to report and none of them are this one.
|
|
805
|
+
"""
|
|
806
|
+
from .adapter_compatibility import adapter_warnings
|
|
807
|
+
|
|
808
|
+
try:
|
|
809
|
+
source_indices = []
|
|
810
|
+
expanded = self.expanded_definition(arguments, source_indices)
|
|
811
|
+
except Exception:
|
|
812
|
+
logger.debug("No adapter warnings available", exc_info=True)
|
|
813
|
+
return []
|
|
814
|
+
return adapter_warnings(
|
|
815
|
+
expanded,
|
|
816
|
+
source_indices,
|
|
817
|
+
written=self.workflow_definition,
|
|
818
|
+
supplied=set(arguments or {}),
|
|
819
|
+
)
|
|
820
|
+
|
|
821
|
+
def slice_past_end_warnings(self, arguments=None):
|
|
822
|
+
"""Every `slice_audio` step whose source's real duration is already
|
|
823
|
+
knowable and whose requested slice reaches past it - valid, padded
|
|
824
|
+
with silence rather than refused, but worth saying before the run
|
|
825
|
+
rather than only after it (#402).
|
|
826
|
+
|
|
827
|
+
Best effort: a definition the schema or the expander refuses has its
|
|
828
|
+
own errors to report and none of them are this one.
|
|
829
|
+
"""
|
|
830
|
+
from .slice_preflight import slice_past_end_warnings
|
|
831
|
+
|
|
832
|
+
try:
|
|
833
|
+
source_indices = []
|
|
834
|
+
expanded = self.expanded_definition(arguments, source_indices)
|
|
835
|
+
except Exception:
|
|
836
|
+
logger.debug("No slice_past_end warnings available", exc_info=True)
|
|
837
|
+
return []
|
|
838
|
+
base_dir = (
|
|
839
|
+
os.path.dirname(os.path.abspath(self.file_spec)) if self.file_spec else None
|
|
840
|
+
)
|
|
841
|
+
return slice_past_end_warnings(expanded, source_indices, base_dir)
|
|
842
|
+
|
|
843
|
+
def shot_span_warnings(self, arguments=None):
|
|
844
|
+
"""Every assessment-probe step (`analyze_shots`, `analyze_seams`,
|
|
845
|
+
`analyze_sync_drift`) whose `shots` argument already reaches past a
|
|
846
|
+
statically-knowable video's real frame count - valid, silently
|
|
847
|
+
clipped to the file rather than refused, but worth saying before the
|
|
848
|
+
run rather than only after it (#425).
|
|
849
|
+
|
|
850
|
+
Best effort: a definition the schema or the expander refuses has its
|
|
851
|
+
own errors to report and none of them are this one.
|
|
852
|
+
"""
|
|
853
|
+
from .shot_span_preflight import shot_span_warnings
|
|
854
|
+
|
|
855
|
+
try:
|
|
856
|
+
source_indices = []
|
|
857
|
+
expanded = self.expanded_definition(arguments, source_indices)
|
|
858
|
+
except Exception:
|
|
859
|
+
logger.debug("No shot_span warnings available", exc_info=True)
|
|
860
|
+
return []
|
|
861
|
+
base_dir = (
|
|
862
|
+
os.path.dirname(os.path.abspath(self.file_spec)) if self.file_spec else None
|
|
863
|
+
)
|
|
864
|
+
return shot_span_warnings(expanded, source_indices, base_dir)
|
|
865
|
+
|
|
866
|
+
def null_variable_argument_warnings(self, arguments=None):
|
|
867
|
+
"""Every required task argument fed by `variable:name` where name's
|
|
868
|
+
value is null - downgraded out of `validation_errors` when
|
|
869
|
+
`arguments` is None (#364), surfaced here so a caller checking the
|
|
870
|
+
document without arguments of its own (save_workflow,
|
|
871
|
+
validate_workflow with no `arguments`) still sees it, just not as a
|
|
872
|
+
reason the document is invalid.
|
|
873
|
+
|
|
874
|
+
Empty once `arguments` is given: at that point the same condition is
|
|
875
|
+
a hard error in `validation_errors`, since a real run or a validate
|
|
876
|
+
call naming its own arguments needed the variable to hold something.
|
|
877
|
+
|
|
878
|
+
Best effort: a definition the schema or the expander refuses has its
|
|
879
|
+
own errors to report and none of them are this one.
|
|
880
|
+
"""
|
|
881
|
+
if arguments is not None:
|
|
882
|
+
return []
|
|
883
|
+
try:
|
|
884
|
+
source_indices = []
|
|
885
|
+
expanded = self.expanded_definition(arguments, source_indices)
|
|
886
|
+
except Exception:
|
|
887
|
+
logger.debug("No null-variable-argument warnings available", exc_info=True)
|
|
888
|
+
return []
|
|
889
|
+
return [
|
|
890
|
+
f"{entry['path']}: {entry['message']}"
|
|
891
|
+
for entry in task_signature_errors(
|
|
892
|
+
expanded, source_indices, self.workflow_definition
|
|
893
|
+
)
|
|
894
|
+
if "variable" in entry
|
|
895
|
+
]
|
|
896
|
+
|
|
897
|
+
def _undeclared_variable_errors(self, arguments=None):
|
|
898
|
+
"""Every 'variable:' reference naming nothing the workflow declares.
|
|
899
|
+
|
|
900
|
+
Fatal rather than a warning: once a workflow has a 'variables'
|
|
901
|
+
block, replace_variables refuses an undeclared reference, so this is
|
|
902
|
+
a run that cannot start. Good caller `arguments` are folded in first,
|
|
903
|
+
and a reference inside one of them is reported under `arguments.`,
|
|
904
|
+
where the caller wrote it.
|
|
905
|
+
"""
|
|
906
|
+
definition = copy.deepcopy(self.workflow_definition)
|
|
907
|
+
variables = definition.get("variables")
|
|
908
|
+
supplied = set()
|
|
909
|
+
if isinstance(variables, dict) and arguments:
|
|
910
|
+
if not argument_errors(definition, arguments):
|
|
911
|
+
set_variables(arguments, variables)
|
|
912
|
+
supplied = set(arguments)
|
|
913
|
+
declared = sorted(variables or {})
|
|
914
|
+
|
|
915
|
+
def where(path):
|
|
916
|
+
head, _, rest = path.partition(".")
|
|
917
|
+
if head == "variables":
|
|
918
|
+
name = rest.split(".", 1)[0].split("[", 1)[0]
|
|
919
|
+
if name in supplied:
|
|
920
|
+
return "arguments." + rest
|
|
921
|
+
return path
|
|
922
|
+
|
|
923
|
+
return [
|
|
924
|
+
{
|
|
925
|
+
"path": where(path),
|
|
926
|
+
"message": (
|
|
927
|
+
f"'variable:{name}' names no declared variable; "
|
|
928
|
+
f"declared: {', '.join(declared) or '<none>'}"
|
|
929
|
+
),
|
|
930
|
+
}
|
|
931
|
+
for path, name in undeclared_variable_references(definition)
|
|
932
|
+
]
|
|
933
|
+
|
|
934
|
+
def validate(self, arguments=None):
|
|
935
|
+
"""Validates workflow definition against JSON schema.
|
|
936
|
+
|
|
937
|
+
Every violation is reported, one per line, so the CLI, the REPL
|
|
938
|
+
and an agent iterating on a draft fix them in one pass rather than
|
|
939
|
+
one per round trip. ``arguments``, when given, are folded in before
|
|
940
|
+
checking - a caller's override (e.g. a content_type-driving variable)
|
|
941
|
+
must be judged as it will actually run, not against the document's
|
|
942
|
+
unsubstituted defaults.
|
|
943
|
+
"""
|
|
944
|
+
logger.debug(f"Validating workflow: {self.name}")
|
|
945
|
+
errors = self.validation_errors(arguments=arguments)
|
|
946
|
+
if errors:
|
|
947
|
+
# message already carries the 'Validation error' prefix
|
|
948
|
+
message = format_validation_errors(errors)
|
|
949
|
+
logger.error(message)
|
|
950
|
+
raise Exception(message)
|
|
951
|
+
logger.debug(f"Workflow {self.name} validated successfully")
|
|
952
|
+
|
|
953
|
+
def _prepare_definition(self, workflow_def, arguments, base_dir):
|
|
954
|
+
"""The definition as a run works from it: constants realized,
|
|
955
|
+
arguments folded into the variables, list entries' own references
|
|
956
|
+
resolved, variable values realized (assets loaded), every
|
|
957
|
+
'variable:' substituted, every for_each expanded, and the seed read
|
|
958
|
+
and coerced. Returns (workflow_def, default_seed) - the seed is
|
|
959
|
+
None when the workflow names none, and the caller decides what
|
|
960
|
+
that means (run() draws one; cache_hits() reports no hits).
|
|
961
|
+
|
|
962
|
+
Shared by run() and cache_hits() so the probe prepares exactly what
|
|
963
|
+
the run prepares - the step cache keys on the realized step, and a
|
|
964
|
+
probe that prepared it differently would answer for a run that
|
|
965
|
+
never happens.
|
|
966
|
+
|
|
967
|
+
Records the steps elision dropped on `self._elided_steps` (#122) -
|
|
968
|
+
the same list run() warns about and writes into the manifest.
|
|
969
|
+
"""
|
|
970
|
+
workflow_id = workflow_def["id"]
|
|
971
|
+
variables = workflow_def.get("variables", None)
|
|
972
|
+
if variables is not None:
|
|
973
|
+
logger.debug(f"Setting variables for workflow: {workflow_id}")
|
|
974
|
+
# a constant is the value a variable declares, so it resolves before
|
|
975
|
+
# anything is converted to the type of that declaration
|
|
976
|
+
realize_constants(variables)
|
|
977
|
+
# first set variable values base don the arguments passed to the workflow
|
|
978
|
+
# these may come form the command line or form a parent workflow
|
|
979
|
+
set_variables(arguments, variables)
|
|
980
|
+
# an entry of a list-valued variable may name another
|
|
981
|
+
# variable; resolve those before anything inside it is
|
|
982
|
+
# realized, so a reference type in an entry is a type name -
|
|
983
|
+
# and before the constraints pass, so an entry written as
|
|
984
|
+
# "variable:tail_len" is a number by the time the rule looks
|
|
985
|
+
variables = resolve_variable_values(variables)
|
|
986
|
+
# A value outside a rule the workflow declares is refused, and
|
|
987
|
+
# one the rule rounds is rounded with a warning saying so -
|
|
988
|
+
# before anything loads, and before substitution puts the value
|
|
989
|
+
# everywhere it is referenced (dw/variable_constraints.py, #96)
|
|
990
|
+
apply_constraints(workflow_def, variables)
|
|
991
|
+
# A (width, height, num_frames)-shaped combination a declared
|
|
992
|
+
# vram_estimate projects past the card 'cost' was measured on -
|
|
993
|
+
# the run-time backstop for a caller that skips
|
|
994
|
+
# validate_workflow, so this raises the same refusal rather
|
|
995
|
+
# than starting a job the decode step was always going to OOM
|
|
996
|
+
# on (dw/vram_estimate.py, #265)
|
|
997
|
+
apply_vram_estimate(workflow_def, variables)
|
|
998
|
+
# realize the variables - explicit references only (asset:,
|
|
999
|
+
# output:, constant:, prompt:, a {media_type, location} dict).
|
|
1000
|
+
# Key-name conventions (an 'image'/'video'/'_type' argument) are
|
|
1001
|
+
# left off here: a variable's own name is not the argument it
|
|
1002
|
+
# will end up filling, so a variable named 'image' fed to a step's
|
|
1003
|
+
# 'video' argument was pre-loaded as a PIL Image before that step
|
|
1004
|
+
# was ever substituted in (#365). The step-level realize_args
|
|
1005
|
+
# passes below apply the conventions under the real argument key
|
|
1006
|
+
realize_args(variables, base_dir, apply_key_conventions=False)
|
|
1007
|
+
## then replace any variable references in the workflow definition with the actual values
|
|
1008
|
+
# replace_variables returns a new structure rather than mutating in
|
|
1009
|
+
# place, so the result must be captured here
|
|
1010
|
+
workflow_def = replace_variables(workflow_def, variables)
|
|
1011
|
+
|
|
1012
|
+
# One ordinary step per entry of every for_each list, before the
|
|
1013
|
+
# seed, the run id and the realized workflow are computed, so
|
|
1014
|
+
# each covers what actually runs. A ForEachError here fails the
|
|
1015
|
+
# run before anything loads
|
|
1016
|
+
# A chain step's `frame_snap` may name the declared constraint
|
|
1017
|
+
# rather than repeating its numbers, so a template states the rule
|
|
1018
|
+
# once (#96)
|
|
1019
|
+
resolve_constraint_references(workflow_def)
|
|
1020
|
+
|
|
1021
|
+
workflow_def = expand_for_each(workflow_def)
|
|
1022
|
+
|
|
1023
|
+
# A step nothing after it reads, and which saves no file, does not
|
|
1024
|
+
# run - after expansion, so a for_each member is judged like any
|
|
1025
|
+
# other step, and before the seed and the run id, so everything
|
|
1026
|
+
# downstream counts the steps that will actually execute
|
|
1027
|
+
# (dw/elision.py, #122)
|
|
1028
|
+
# The definition as written is passed too: a step dropped because a
|
|
1029
|
+
# caller replaced the variable that read it was elided on purpose,
|
|
1030
|
+
# and says so, rather than being reported as a suspected typo (#157)
|
|
1031
|
+
self._elided_steps = elide_definition(workflow_def, self.workflow_definition)
|
|
1032
|
+
|
|
1033
|
+
# Set up random seed for reproducibility. Resolved lazily - as a
|
|
1034
|
+
# dict.get default, torch.seed() would run on every call and reseed
|
|
1035
|
+
# the global RNG even when the workflow names an explicit seed
|
|
1036
|
+
default_seed = workflow_def.get("seed")
|
|
1037
|
+
# The schema lets 'seed' be a string so it can hold a 'variable:'
|
|
1038
|
+
# reference, which the substitution above has already resolved -
|
|
1039
|
+
# but a variable overridden from the command line arrives as a
|
|
1040
|
+
# string whenever the workflow declared no integer default to
|
|
1041
|
+
# coerce against, and manual_seed would fail deep inside the run
|
|
1042
|
+
if isinstance(default_seed, str):
|
|
1043
|
+
try:
|
|
1044
|
+
default_seed = int(default_seed)
|
|
1045
|
+
except ValueError:
|
|
1046
|
+
raise ValueError(
|
|
1047
|
+
f"Workflow {workflow_id} seed must be an integer, "
|
|
1048
|
+
f"got {default_seed!r}"
|
|
1049
|
+
)
|
|
1050
|
+
workflow_def["seed"] = default_seed
|
|
1051
|
+
return workflow_def, default_seed
|
|
1052
|
+
|
|
1053
|
+
def _cache_lookup(
|
|
1054
|
+
self,
|
|
1055
|
+
workflow_id,
|
|
1056
|
+
steps,
|
|
1057
|
+
index,
|
|
1058
|
+
step_data,
|
|
1059
|
+
step_seed,
|
|
1060
|
+
hits_this_run,
|
|
1061
|
+
cache_enabled,
|
|
1062
|
+
):
|
|
1063
|
+
"""Whether the step cache serves step `index`, as (cached_result or
|
|
1064
|
+
None, the step_data snapshot the entry is keyed on or None, whether
|
|
1065
|
+
a later step still reads this one's result, the names later steps
|
|
1066
|
+
still reference). Shared by run() and cache_hits() - see
|
|
1067
|
+
_prepare_definition for why.
|
|
1068
|
+
"""
|
|
1069
|
+
# What later steps still read, which decides both whether this
|
|
1070
|
+
# step's result has to be kept alive after the step (release_unreferenced_results
|
|
1071
|
+
# at the bottom of the run loop) and whether a cached entry that
|
|
1072
|
+
# kept none can serve this run
|
|
1073
|
+
remaining_refs = referenced_result_names(steps[index + 1 :])
|
|
1074
|
+
result_needed = index == len(steps) - 1 or any(
|
|
1075
|
+
reference_resolves_to(ref, step_data["name"]) for ref in remaining_refs
|
|
1076
|
+
)
|
|
1077
|
+
# create_step_action (and the pipeline load it triggers) mutates
|
|
1078
|
+
# step_data in place - injecting a "generator" key - so the cache
|
|
1079
|
+
# must key off a snapshot taken before that happens, and that same
|
|
1080
|
+
# snapshot must be reused for the put() later. Caching off the live,
|
|
1081
|
+
# later-mutated step_data would make every step's dict keys diverge
|
|
1082
|
+
# from a freshly deep-copied future run's step_data, so get() would
|
|
1083
|
+
# never match again after the first run.
|
|
1084
|
+
# A sub-workflow step is never cacheable: its files roll up from the
|
|
1085
|
+
# child's own manifest, which a hit does not rebuild.
|
|
1086
|
+
is_cacheable = "workflow" not in step_data and cache_enabled
|
|
1087
|
+
# The last step of a composed child whose parent does the saving
|
|
1088
|
+
# (#92) - its files are the parent step's, written once, under the
|
|
1089
|
+
# parent's name and subfolder
|
|
1090
|
+
parent_saves_this = self._final_save_owned_by_parent and index == len(steps) - 1
|
|
1091
|
+
step_data_snapshot = None
|
|
1092
|
+
if is_cacheable:
|
|
1093
|
+
try:
|
|
1094
|
+
step_data_snapshot = copy.deepcopy(step_data)
|
|
1095
|
+
if parent_saves_this:
|
|
1096
|
+
# Keyed apart from the same step run standalone: this
|
|
1097
|
+
# entry's result was never saved here, so a standalone
|
|
1098
|
+
# hit on it would report no files
|
|
1099
|
+
step_data_snapshot["__saved_by_parent__"] = True
|
|
1100
|
+
except Exception as ex:
|
|
1101
|
+
# A realized argument that cannot be deep-copied (an open
|
|
1102
|
+
# handle, a live model object) just means this step is not
|
|
1103
|
+
# cacheable - never a failed run
|
|
1104
|
+
logger.debug(
|
|
1105
|
+
f"Step '{step_data['name']}' arguments are not copyable "
|
|
1106
|
+
f"({ex}) - skipping the step cache for it"
|
|
1107
|
+
)
|
|
1108
|
+
is_cacheable = False
|
|
1109
|
+
if not is_cacheable:
|
|
1110
|
+
return None, None, result_needed, remaining_refs
|
|
1111
|
+
cached_result = step_cache.get(
|
|
1112
|
+
workflow_id,
|
|
1113
|
+
step_data_snapshot,
|
|
1114
|
+
step_seed,
|
|
1115
|
+
hits_this_run,
|
|
1116
|
+
# The root, not this run's directory: a hit reports the earlier
|
|
1117
|
+
# run's files and writes nothing new, so keying on a directory
|
|
1118
|
+
# that is new every run would mean the cache could never hit
|
|
1119
|
+
# again. What the root still guards is a run redirected
|
|
1120
|
+
# somewhere else, where the earlier files are not what the
|
|
1121
|
+
# caller asked for
|
|
1122
|
+
self.output_dir,
|
|
1123
|
+
needs_result=result_needed,
|
|
1124
|
+
)
|
|
1125
|
+
return cached_result, step_data_snapshot, result_needed, remaining_refs
|
|
1126
|
+
|
|
1127
|
+
def cache_hits(self, arguments):
|
|
1128
|
+
"""The steps the step cache would serve for a run with `arguments`,
|
|
1129
|
+
in step order - what the plan reports as cached_steps (#85).
|
|
1130
|
+
|
|
1131
|
+
Prepares the definition exactly as run() does and asks the cache the
|
|
1132
|
+
question run() asks, step by step with the hits so far, and executes
|
|
1133
|
+
nothing: no run directory, no events, no pipeline. An unseeded
|
|
1134
|
+
workflow has no cache, so it answers [] without asking.
|
|
1135
|
+
"""
|
|
1136
|
+
output_root_token = activate_output_root(self.output_dir)
|
|
1137
|
+
try:
|
|
1138
|
+
workflow_def = copy.deepcopy(self.workflow_definition)
|
|
1139
|
+
workflow_id = workflow_def["id"]
|
|
1140
|
+
base_dir = (
|
|
1141
|
+
os.path.dirname(os.path.abspath(self.file_spec))
|
|
1142
|
+
if self.file_spec
|
|
1143
|
+
else None
|
|
1144
|
+
)
|
|
1145
|
+
workflow_def, default_seed = self._prepare_definition(
|
|
1146
|
+
workflow_def, arguments or {}, base_dir
|
|
1147
|
+
)
|
|
1148
|
+
if default_seed is None or not self._cache_enabled_by_parent:
|
|
1149
|
+
return []
|
|
1150
|
+
steps = workflow_def.get("steps", [])
|
|
1151
|
+
realize_args(steps, base_dir)
|
|
1152
|
+
hits_this_run = set()
|
|
1153
|
+
hits = []
|
|
1154
|
+
for index, step_data in enumerate(steps):
|
|
1155
|
+
step_seed = step_data.get("seed", default_seed)
|
|
1156
|
+
cached_result, _, _, _ = self._cache_lookup(
|
|
1157
|
+
workflow_id,
|
|
1158
|
+
steps,
|
|
1159
|
+
index,
|
|
1160
|
+
step_data,
|
|
1161
|
+
step_seed,
|
|
1162
|
+
hits_this_run,
|
|
1163
|
+
True,
|
|
1164
|
+
)
|
|
1165
|
+
if cached_result is not None:
|
|
1166
|
+
hits_this_run.add(step_data["name"])
|
|
1167
|
+
hits.append(step_data["name"])
|
|
1168
|
+
return hits
|
|
1169
|
+
finally:
|
|
1170
|
+
deactivate_output_root(output_root_token)
|
|
1171
|
+
|
|
1172
|
+
def run(
|
|
1173
|
+
self, arguments, previous_pipelines=None, context=None, prior_step_keys=None
|
|
1174
|
+
):
|
|
1175
|
+
"""
|
|
1176
|
+
Executes the workflow by:
|
|
1177
|
+
1. Processing variables
|
|
1178
|
+
2. Setting up random seed
|
|
1179
|
+
3. Running each step in sequence
|
|
1180
|
+
4. Managing results between steps
|
|
1181
|
+
|
|
1182
|
+
An explicit RunContext receives progress events and can cancel the
|
|
1183
|
+
run; without one, the ambient context is reused (a sub-workflow
|
|
1184
|
+
reports into its parent's run) or a no-op context is created.
|
|
1185
|
+
Saved file paths accumulate in self.manifest, one entry per step.
|
|
1186
|
+
"""
|
|
1187
|
+
run_context = context or current_context() or RunContext()
|
|
1188
|
+
context_token = activate_context(run_context)
|
|
1189
|
+
# Depth-counted: a sub-workflow shares its parent's RunContext, so the
|
|
1190
|
+
# phase-stall watchdog starts once on the outermost run() and stops
|
|
1191
|
+
# once that outermost call's finally below runs, not on every nested
|
|
1192
|
+
# sub-workflow call
|
|
1193
|
+
run_context.enter_run()
|
|
1194
|
+
# 'output:' references resolve against the directory this run was
|
|
1195
|
+
# told to write to - the root, not this run's own subdirectory, since
|
|
1196
|
+
# what they name is what an earlier run left there
|
|
1197
|
+
output_root_token = activate_output_root(self.output_dir)
|
|
1198
|
+
# Step name -> cache key for this run, so release_pipeline and
|
|
1199
|
+
# pipeline_reference still address pipelines by the step that made them
|
|
1200
|
+
self._pipeline_keys_by_step = {}
|
|
1201
|
+
# Last run's step->key map: a redefined step's old model is evicted
|
|
1202
|
+
# BEFORE its replacement loads, or the transition holds both at once
|
|
1203
|
+
self._prior_step_keys = prior_step_keys or {}
|
|
1204
|
+
self.manifest = []
|
|
1205
|
+
# What elision dropped this run, filled by _prepare_definition and
|
|
1206
|
+
# read by the warning pass and the manifest (#122)
|
|
1207
|
+
self._elided_steps = []
|
|
1208
|
+
# Overwritten on the way out of the try below - a run that leaves
|
|
1209
|
+
# this alone died on an exception the manifest should say so about
|
|
1210
|
+
status = "failed"
|
|
1211
|
+
run_id = None
|
|
1212
|
+
# The seed this run actually used, which is not self.workflow_definition's:
|
|
1213
|
+
# run() works on a deep copy, and a workflow naming no seed draws a random
|
|
1214
|
+
# one into that copy. Recording the original would write null into the
|
|
1215
|
+
# manifest of every seedless run and lose the only record of what produced
|
|
1216
|
+
# its files - the seed is what makes a run repeatable
|
|
1217
|
+
resolved_seed = None
|
|
1218
|
+
# What the run recorded about itself, read by _write_run_manifest in
|
|
1219
|
+
# the finally below - initialized here so a failure before the run
|
|
1220
|
+
# directory exists still writes a well-formed manifest
|
|
1221
|
+
realized_name = None
|
|
1222
|
+
annotations = {"prompts": [], "sub_workflows": {}}
|
|
1223
|
+
started_at = datetime.now(timezone.utc).isoformat()
|
|
1224
|
+
try:
|
|
1225
|
+
# CRITICAL: Work on a copy to avoid mutating the original workflow definition
|
|
1226
|
+
# This allows the workflow to be run multiple times with different arguments
|
|
1227
|
+
workflow_def = copy.deepcopy(self.workflow_definition)
|
|
1228
|
+
|
|
1229
|
+
workflow_id = workflow_def["id"]
|
|
1230
|
+
logger.debug(f"Processing workflow: {workflow_id}")
|
|
1231
|
+
|
|
1232
|
+
# File paths in workflows are relative to the workflow file
|
|
1233
|
+
base_dir = (
|
|
1234
|
+
os.path.dirname(os.path.abspath(self.file_spec))
|
|
1235
|
+
if self.file_spec
|
|
1236
|
+
else None
|
|
1237
|
+
)
|
|
1238
|
+
|
|
1239
|
+
workflow_def, default_seed = self._prepare_definition(
|
|
1240
|
+
workflow_def, arguments, base_dir
|
|
1241
|
+
)
|
|
1242
|
+
# An adapter whose name says nothing about what it was trained
|
|
1243
|
+
# for cannot be checked, and a run that started from the CLI or
|
|
1244
|
+
# from a rerun never passed the validate route (#155)
|
|
1245
|
+
warn_adapters(workflow_def)
|
|
1246
|
+
# Said out loud before anything loads: a step that vanishes
|
|
1247
|
+
# because a reference to it is misspelled would otherwise show
|
|
1248
|
+
# up only as a different picture (#122)
|
|
1249
|
+
warn_elided(self._elided_steps)
|
|
1250
|
+
# A workflow that names no seed gets a fresh one every run, so no
|
|
1251
|
+
# step's cache entry can ever match again - skip the cache
|
|
1252
|
+
# wholesale rather than deep-copying every step's realized images
|
|
1253
|
+
# and pinning every Result for a hit that cannot happen
|
|
1254
|
+
cache_enabled_this_run = (
|
|
1255
|
+
self._cache_enabled_by_parent and default_seed is not None
|
|
1256
|
+
)
|
|
1257
|
+
# create_step_action hands this down to a sub-workflow: it injects
|
|
1258
|
+
# the parent's seed into a child that names none, so a child of a
|
|
1259
|
+
# seedless parent would otherwise look seeded - and cacheable -
|
|
1260
|
+
# while its seed still changes every run
|
|
1261
|
+
self._cache_enabled_this_run = cache_enabled_this_run
|
|
1262
|
+
if default_seed is None:
|
|
1263
|
+
# OS entropy rather than torch or random, so a process that
|
|
1264
|
+
# seeded either for reproducibility is not disturbed. Bounded
|
|
1265
|
+
# to 53 bits rather than the 64 torch allows: the seed is
|
|
1266
|
+
# embedded in the image, the manifest and the realized
|
|
1267
|
+
# workflow as JSON, and a browser reads every integer as a
|
|
1268
|
+
# double - a seed that changed on the way through would be a
|
|
1269
|
+
# seed nobody can reproduce
|
|
1270
|
+
default_seed = secrets.randbits(SEED_BITS)
|
|
1271
|
+
workflow_def["seed"] = default_seed
|
|
1272
|
+
resolved_seed = default_seed
|
|
1273
|
+
|
|
1274
|
+
# One execution, one directory - opened here, after variable
|
|
1275
|
+
# substitution and the seed have settled, so the run's identity
|
|
1276
|
+
# covers what actually ran rather than what was written down. A
|
|
1277
|
+
# sub-workflow inherits the parent's and never opens its own
|
|
1278
|
+
started_at = datetime.now(timezone.utc).isoformat()
|
|
1279
|
+
run_id = None
|
|
1280
|
+
if not self._run_dir_inherited:
|
|
1281
|
+
if output_layout() == FLAT_LAYOUT:
|
|
1282
|
+
self._run_dir = None
|
|
1283
|
+
else:
|
|
1284
|
+
run_id = new_run_id(
|
|
1285
|
+
{"workflow": workflow_def, "arguments": arguments}
|
|
1286
|
+
)
|
|
1287
|
+
self._run_dir = run_directory(
|
|
1288
|
+
self.output_dir, self.file_spec, workflow_id, run_id
|
|
1289
|
+
)
|
|
1290
|
+
# The run's ordinal among this workflow's runs, taken
|
|
1291
|
+
# once here and carried into the manifest. Assigning it
|
|
1292
|
+
# at run time rather than deriving it when the gallery
|
|
1293
|
+
# asks is what lets a sibling be deleted without
|
|
1294
|
+
# renumbering the runs that outlive it
|
|
1295
|
+
self._run_version = assign_run_version(
|
|
1296
|
+
self.output_dir,
|
|
1297
|
+
workflow_identity(self.file_spec, workflow_id),
|
|
1298
|
+
)
|
|
1299
|
+
logger.debug(
|
|
1300
|
+
f"Run directory: {self._run_dir} (v{self._run_version})"
|
|
1301
|
+
)
|
|
1302
|
+
|
|
1303
|
+
# The record of what actually ran, written before the first step
|
|
1304
|
+
# so a crash or a cancel still leaves it. A sub-workflow inherits
|
|
1305
|
+
# the parent's directory and writes none of its own, as with the
|
|
1306
|
+
# manifest, and the flat layout has no directory to write into
|
|
1307
|
+
if self._run_dir and not self._run_dir_inherited:
|
|
1308
|
+
try:
|
|
1309
|
+
realized, annotations = realize_workflow(
|
|
1310
|
+
self.workflow_definition,
|
|
1311
|
+
arguments,
|
|
1312
|
+
default_seed,
|
|
1313
|
+
base_dir=base_dir,
|
|
1314
|
+
output_root=self.output_dir,
|
|
1315
|
+
workflow_dir=self.workflow_dir,
|
|
1316
|
+
)
|
|
1317
|
+
if write_realized_workflow(self._run_dir, realized):
|
|
1318
|
+
realized_name = REALIZED_FILE_NAME
|
|
1319
|
+
except Exception as e:
|
|
1320
|
+
# Never fatal: the record is worth less than the run
|
|
1321
|
+
logger.warning(f"Could not realize workflow {workflow_id}: {e}")
|
|
1322
|
+
|
|
1323
|
+
# A manifest now, rewritten in full when the run ends: the
|
|
1324
|
+
# version held only in memory until then was lost to a hard
|
|
1325
|
+
# kill, and a second process opening a run of this workflow
|
|
1326
|
+
# meanwhile could not see it and took the same number
|
|
1327
|
+
self._write_run_manifest(
|
|
1328
|
+
run_id,
|
|
1329
|
+
"running",
|
|
1330
|
+
started_at,
|
|
1331
|
+
arguments,
|
|
1332
|
+
resolved_seed,
|
|
1333
|
+
realized_name,
|
|
1334
|
+
annotations,
|
|
1335
|
+
)
|
|
1336
|
+
|
|
1337
|
+
# Which run this is, so a server job can find the directory
|
|
1338
|
+
# it wrote. Emitted even when the realized file did not land:
|
|
1339
|
+
# the manifest is still there, and so are the files
|
|
1340
|
+
run_context.emit(
|
|
1341
|
+
"run_start",
|
|
1342
|
+
run_id=run_id,
|
|
1343
|
+
version=self._run_version,
|
|
1344
|
+
identity=workflow_identity(self.file_spec, workflow_id),
|
|
1345
|
+
run_dir=os.path.relpath(self._run_dir, self.output_dir).replace(
|
|
1346
|
+
os.sep, "/"
|
|
1347
|
+
),
|
|
1348
|
+
)
|
|
1349
|
+
|
|
1350
|
+
# Initialize collections for sharing state between steps
|
|
1351
|
+
# Stores results from each step, and remembers the names of
|
|
1352
|
+
# steps whose results have since been released
|
|
1353
|
+
results = StepResults()
|
|
1354
|
+
shared_components = {} # Shared resources between steps
|
|
1355
|
+
|
|
1356
|
+
# Use provided pipelines cache or create new dict
|
|
1357
|
+
# This allows pipeline reuse across multiple workflow runs
|
|
1358
|
+
if previous_pipelines is None:
|
|
1359
|
+
pipelines = {}
|
|
1360
|
+
logger.debug("Starting with empty pipeline cache")
|
|
1361
|
+
else:
|
|
1362
|
+
pipelines = previous_pipelines
|
|
1363
|
+
logger.debug(f"Reusing pipeline cache with {len(pipelines)} pipelines")
|
|
1364
|
+
|
|
1365
|
+
last_result = None # Final result is the workflow return value
|
|
1366
|
+
|
|
1367
|
+
# realize any arguments for the steps, i.e. load images etc
|
|
1368
|
+
# that are referenced directly in the step
|
|
1369
|
+
steps = workflow_def.get("steps", [])
|
|
1370
|
+
|
|
1371
|
+
if not steps:
|
|
1372
|
+
logger.warning(f"Workflow {workflow_id} has no steps defined")
|
|
1373
|
+
status = "completed"
|
|
1374
|
+
return []
|
|
1375
|
+
|
|
1376
|
+
realize_args(steps, base_dir)
|
|
1377
|
+
|
|
1378
|
+
# The key each pipeline step of THIS run loads under, computed
|
|
1379
|
+
# from the same realized dicts create_step_action hashes, so the
|
|
1380
|
+
# two agree. This is what "still shared" means there: a key
|
|
1381
|
+
# another running step maps to NOW - not the key it mapped to
|
|
1382
|
+
# last run (every step sharing a changed model variable has the
|
|
1383
|
+
# old key as its prior key and none has it as its current one),
|
|
1384
|
+
# and not a key some step of a past, unrelated workflow left in
|
|
1385
|
+
# the cross-job _prior_step_keys map
|
|
1386
|
+
self._running_pipeline_keys = {
|
|
1387
|
+
step_data["name"]: pipeline_cache_key(step_data["pipeline"])
|
|
1388
|
+
for step_data in steps
|
|
1389
|
+
if "pipeline" in step_data
|
|
1390
|
+
}
|
|
1391
|
+
|
|
1392
|
+
run_context.emit(
|
|
1393
|
+
"workflow_start",
|
|
1394
|
+
workflow=workflow_id,
|
|
1395
|
+
total_steps=len(steps),
|
|
1396
|
+
steps=[step_data["name"] for step_data in steps],
|
|
1397
|
+
seed=default_seed,
|
|
1398
|
+
)
|
|
1399
|
+
|
|
1400
|
+
# Step name -> whether that step's result this run came from the
|
|
1401
|
+
# cache, so a step that reads another step's result can tell
|
|
1402
|
+
# whether its own inputs are still all cache-fresh
|
|
1403
|
+
hits_this_run = set()
|
|
1404
|
+
|
|
1405
|
+
# Execute each step in sequence
|
|
1406
|
+
for i, step_data in enumerate(steps):
|
|
1407
|
+
run_context.check_cancelled()
|
|
1408
|
+
logger.debug(f"Running step {i + 1}/{len(steps)}: {step_data['name']}")
|
|
1409
|
+
run_context.emit(
|
|
1410
|
+
"step_start",
|
|
1411
|
+
workflow=workflow_id,
|
|
1412
|
+
step=step_data["name"],
|
|
1413
|
+
index=i,
|
|
1414
|
+
total_steps=len(steps),
|
|
1415
|
+
**self._parent_progress_fields(),
|
|
1416
|
+
)
|
|
1417
|
+
|
|
1418
|
+
# Seeds resolve most-specific-first: pipeline > step > workflow
|
|
1419
|
+
step_seed = step_data.get("seed", default_seed)
|
|
1420
|
+
|
|
1421
|
+
step = Step(
|
|
1422
|
+
step_data,
|
|
1423
|
+
step_seed,
|
|
1424
|
+
self.workflow_definition,
|
|
1425
|
+
consumed_by_normalizer=normalized_downstream(
|
|
1426
|
+
steps[i + 1 :], step_data["name"]
|
|
1427
|
+
),
|
|
1428
|
+
)
|
|
1429
|
+
|
|
1430
|
+
cached_result, step_data_snapshot, result_needed, remaining_refs = (
|
|
1431
|
+
self._cache_lookup(
|
|
1432
|
+
workflow_id,
|
|
1433
|
+
steps,
|
|
1434
|
+
i,
|
|
1435
|
+
step_data,
|
|
1436
|
+
step_seed,
|
|
1437
|
+
hits_this_run,
|
|
1438
|
+
cache_enabled_this_run,
|
|
1439
|
+
)
|
|
1440
|
+
)
|
|
1441
|
+
is_cacheable = step_data_snapshot is not None
|
|
1442
|
+
# The last step of a composed child whose parent does the
|
|
1443
|
+
# saving (#92) - its files are written once, by the parent
|
|
1444
|
+
parent_saves_this = (
|
|
1445
|
+
self._final_save_owned_by_parent and i == len(steps) - 1
|
|
1446
|
+
)
|
|
1447
|
+
|
|
1448
|
+
# A hit skips the step's work, never its bookkeeping:
|
|
1449
|
+
# create_step_action is the only place that touches the
|
|
1450
|
+
# step's pipeline (the worker evicts every pipeline a run did
|
|
1451
|
+
# not touch), republishes a cached pipeline's
|
|
1452
|
+
# shared_components for a later reusing step, and records the
|
|
1453
|
+
# step's pipeline key for release_pipeline and
|
|
1454
|
+
# pipeline_reference to address it by
|
|
1455
|
+
step_action = self.create_step_action(
|
|
1456
|
+
step_data,
|
|
1457
|
+
shared_components,
|
|
1458
|
+
pipelines,
|
|
1459
|
+
step_seed,
|
|
1460
|
+
get_device(),
|
|
1461
|
+
)
|
|
1462
|
+
if isinstance(step_action, Workflow):
|
|
1463
|
+
# The child reports into this run's counter rather than
|
|
1464
|
+
# its own, and a grandchild reports into the same one
|
|
1465
|
+
step_action._parent_progress = self._parent_progress or {
|
|
1466
|
+
"step": step_data["name"],
|
|
1467
|
+
"index": i,
|
|
1468
|
+
"total_steps": len(steps),
|
|
1469
|
+
}
|
|
1470
|
+
# Only when the parent's own result would write
|
|
1471
|
+
# something: a result block that names no content_type,
|
|
1472
|
+
# or says save: false, saves nothing, and suppressing
|
|
1473
|
+
# the child's save for it would lose the artifact
|
|
1474
|
+
parent_result = step_data.get("result")
|
|
1475
|
+
step_action._final_save_owned_by_parent = bool(
|
|
1476
|
+
isinstance(parent_result, dict)
|
|
1477
|
+
and parent_result.get("content_type")
|
|
1478
|
+
and parent_result.get("save", True)
|
|
1479
|
+
)
|
|
1480
|
+
reused = cached_result is not None
|
|
1481
|
+
if reused:
|
|
1482
|
+
logger.info(f"Step '{step.name}' unchanged - reusing cached result")
|
|
1483
|
+
result = cached_result
|
|
1484
|
+
saved_files = result.saved_files
|
|
1485
|
+
hits_this_run.add(step.name)
|
|
1486
|
+
else:
|
|
1487
|
+
result = step.run(results, pipelines, step_action)
|
|
1488
|
+
|
|
1489
|
+
# A sub-workflow's saves land in the child's manifest - read it
|
|
1490
|
+
# here, before the release below may drop the child
|
|
1491
|
+
sub_manifest = (
|
|
1492
|
+
list(getattr(step_action, "manifest", []))
|
|
1493
|
+
if isinstance(step_action, Workflow)
|
|
1494
|
+
else []
|
|
1495
|
+
)
|
|
1496
|
+
|
|
1497
|
+
# A released pipeline frees its memory for later steps - the
|
|
1498
|
+
# alternative on a card that cannot hold two models is offloading
|
|
1499
|
+
# everything, which taxes every run to survive one transition.
|
|
1500
|
+
# Before the write, not after: the result is already in host
|
|
1501
|
+
# memory and saving never touches the pipeline, so a release
|
|
1502
|
+
# that waited for the write would hold ~10 GB on the device
|
|
1503
|
+
# through the longest phase of a video step. The loop's own
|
|
1504
|
+
# locals are the last references to this step's action, so
|
|
1505
|
+
# clearing that is part of the release - a popped pipeline this
|
|
1506
|
+
# frame still holds is not freed, and it would otherwise stay
|
|
1507
|
+
# resident through the next step's load, which is exactly when
|
|
1508
|
+
# both models would be in memory at once
|
|
1509
|
+
if step_data.get("release_pipeline", False):
|
|
1510
|
+
logger.info(f"Releasing pipeline for step: {step.name}")
|
|
1511
|
+
before = _allocated_mb()
|
|
1512
|
+
pipelines.pop(self._pipeline_keys_by_step.get(step.name), None)
|
|
1513
|
+
step_action = None
|
|
1514
|
+
gc.collect()
|
|
1515
|
+
empty_device_cache()
|
|
1516
|
+
# The device cache is not the only one the release fills:
|
|
1517
|
+
# the pinned-host staging buffers the pipeline offloaded
|
|
1518
|
+
# through and the heap arenas its weights were read into
|
|
1519
|
+
# stay in this process's RSS until they are handed back,
|
|
1520
|
+
# which otherwise waits for the end of the job - ~10 GB
|
|
1521
|
+
# held through every step after the release (#368)
|
|
1522
|
+
_release_host_caches(step.name)
|
|
1523
|
+
# Say so on the event stream. The release is otherwise
|
|
1524
|
+
# invisible to a consumer: it sits inside the sub-second
|
|
1525
|
+
# window between a step's generation and its files
|
|
1526
|
+
# appearing, which is too narrow to catch by polling
|
|
1527
|
+
# get_memory, and it is exactly the ordering this event
|
|
1528
|
+
# exists to make readable (it precedes the step's
|
|
1529
|
+
# step_end, and on a released card the figures show the
|
|
1530
|
+
# drop rather than implying it)
|
|
1531
|
+
run_context.emit(
|
|
1532
|
+
"pipeline_released",
|
|
1533
|
+
workflow=workflow_id,
|
|
1534
|
+
step=step.name,
|
|
1535
|
+
index=i,
|
|
1536
|
+
gpu_memory_allocated_mb=_allocated_mb(),
|
|
1537
|
+
gpu_memory_allocated_before_mb=before,
|
|
1538
|
+
)
|
|
1539
|
+
|
|
1540
|
+
if not reused:
|
|
1541
|
+
saved_files = (
|
|
1542
|
+
[]
|
|
1543
|
+
if parent_saves_this
|
|
1544
|
+
else result.save(
|
|
1545
|
+
self.step_output_dir(step_data),
|
|
1546
|
+
self.step_save_name(workflow_id, step.name, i),
|
|
1547
|
+
)
|
|
1548
|
+
)
|
|
1549
|
+
if is_cacheable:
|
|
1550
|
+
step_cache.put(
|
|
1551
|
+
workflow_id,
|
|
1552
|
+
step_data_snapshot,
|
|
1553
|
+
step_seed,
|
|
1554
|
+
result,
|
|
1555
|
+
self.output_dir,
|
|
1556
|
+
retain_result=result_needed,
|
|
1557
|
+
)
|
|
1558
|
+
|
|
1559
|
+
last_result = result
|
|
1560
|
+
results[step.name] = result
|
|
1561
|
+
# 'reused' marks files an earlier run wrote and this one only
|
|
1562
|
+
# republished, so nothing downstream (job_for_file, the
|
|
1563
|
+
# gallery) credits this run with writing them
|
|
1564
|
+
subfolder = step_subfolder(step_data)
|
|
1565
|
+
selected = selected_field(step_data, result.selected)
|
|
1566
|
+
manifest_entry = {
|
|
1567
|
+
"step": step.name,
|
|
1568
|
+
"files": saved_files,
|
|
1569
|
+
"subfolder": subfolder,
|
|
1570
|
+
}
|
|
1571
|
+
if reused:
|
|
1572
|
+
manifest_entry["reused"] = True
|
|
1573
|
+
if selected is not None:
|
|
1574
|
+
manifest_entry["selected"] = selected
|
|
1575
|
+
# Where each joined shot sits in the file, named by the
|
|
1576
|
+
# step's own `videos` references (dw/shots.py)
|
|
1577
|
+
shots = step_shots(
|
|
1578
|
+
getattr(result, "saved_shots", None),
|
|
1579
|
+
saved_files,
|
|
1580
|
+
step_data.get("task", {}).get("arguments", {}).get("videos"),
|
|
1581
|
+
)
|
|
1582
|
+
if shots:
|
|
1583
|
+
manifest_entry["shots"] = shots
|
|
1584
|
+
# No entry at all for a step the parent saves for: the
|
|
1585
|
+
# parent's own entry names the same files, under the step
|
|
1586
|
+
# name the caller wrote (#92)
|
|
1587
|
+
if not parent_saves_this:
|
|
1588
|
+
self.manifest.append(manifest_entry)
|
|
1589
|
+
# roll the child's saves up so job history and the gallery see
|
|
1590
|
+
# every file
|
|
1591
|
+
self.manifest.extend(sub_manifest)
|
|
1592
|
+
step_end_data = {"files": saved_files, "subfolder": subfolder}
|
|
1593
|
+
if reused:
|
|
1594
|
+
step_end_data["reused"] = True
|
|
1595
|
+
if selected is not None:
|
|
1596
|
+
step_end_data["selected"] = selected
|
|
1597
|
+
if shots:
|
|
1598
|
+
step_end_data["shots"] = shots
|
|
1599
|
+
run_context.emit(
|
|
1600
|
+
"step_end",
|
|
1601
|
+
workflow=workflow_id,
|
|
1602
|
+
step=step.name,
|
|
1603
|
+
index=i,
|
|
1604
|
+
total_steps=len(steps),
|
|
1605
|
+
**self._parent_progress_fields(),
|
|
1606
|
+
**step_end_data,
|
|
1607
|
+
)
|
|
1608
|
+
logger.debug(f"Step {step.name} completed with result: {result}")
|
|
1609
|
+
|
|
1610
|
+
# Release results no later step references - saved to disk
|
|
1611
|
+
# already, and last_result keeps the workflow's return value
|
|
1612
|
+
release_unreferenced_results(results, remaining_refs)
|
|
1613
|
+
|
|
1614
|
+
# The loop's own locals are the last references to this step's
|
|
1615
|
+
# action and result - anything they still hold would stay
|
|
1616
|
+
# resident through the next step's load
|
|
1617
|
+
step_action = None
|
|
1618
|
+
result = None
|
|
1619
|
+
|
|
1620
|
+
# Task models are cached for the life of the process - the cache
|
|
1621
|
+
# exists so a step's cartesian product loads its model once, and
|
|
1622
|
+
# nothing else evicts it. A prompt-expanding language model
|
|
1623
|
+
# feeding a generation step would otherwise hold its weights on
|
|
1624
|
+
# the device for the whole run
|
|
1625
|
+
if step_data.get("release_models", False):
|
|
1626
|
+
logger.info(f"Releasing task models for step: {step.name}")
|
|
1627
|
+
clear_model_cache()
|
|
1628
|
+
gc.collect()
|
|
1629
|
+
_release_host_caches(step.name)
|
|
1630
|
+
|
|
1631
|
+
# Cleanup between steps (but keep pipelines loaded). Returning
|
|
1632
|
+
# cached blocks to the device lets the next step's differently
|
|
1633
|
+
# shaped allocations use them
|
|
1634
|
+
gc.collect()
|
|
1635
|
+
empty_device_cache()
|
|
1636
|
+
|
|
1637
|
+
logger.debug(f"Workflow {workflow_id} completed successfully")
|
|
1638
|
+
run_context.emit(
|
|
1639
|
+
"workflow_end", workflow=workflow_id, manifest=self.manifest
|
|
1640
|
+
)
|
|
1641
|
+
# Return only the last step's results for child workflows
|
|
1642
|
+
status = "completed"
|
|
1643
|
+
return last_result.result_list if last_result is not None else []
|
|
1644
|
+
|
|
1645
|
+
except WorkflowCancelled:
|
|
1646
|
+
# The user asked for this - report it without an error traceback
|
|
1647
|
+
workflow_id = self.workflow_definition.get("id", "unknown")
|
|
1648
|
+
logger.info(f"Workflow {workflow_id} cancelled")
|
|
1649
|
+
status = "cancelled"
|
|
1650
|
+
raise
|
|
1651
|
+
except (SecurityError, PathTraversalError, InvalidInputError) as e:
|
|
1652
|
+
# Security validation failures - these should fail fast, without the
|
|
1653
|
+
# traceback noise of the general handler
|
|
1654
|
+
workflow_id = self.workflow_definition.get("id", "unknown")
|
|
1655
|
+
logger.error(f"Security error in workflow {workflow_id}: {e}")
|
|
1656
|
+
raise
|
|
1657
|
+
except Exception as e:
|
|
1658
|
+
# One log line with the full traceback - the step already logged its
|
|
1659
|
+
# own context, and every clause here did the same log-and-reraise
|
|
1660
|
+
workflow_id = self.workflow_definition.get("id", "unknown")
|
|
1661
|
+
logger.error(
|
|
1662
|
+
f"{type(e).__name__} in workflow {workflow_id}: {e}", exc_info=True
|
|
1663
|
+
)
|
|
1664
|
+
raise
|
|
1665
|
+
finally:
|
|
1666
|
+
# Recorded even for a run that failed part way: the files it did
|
|
1667
|
+
# write are on disk either way, and what produced them is exactly
|
|
1668
|
+
# what a failed run needs to explain itself
|
|
1669
|
+
if self._run_dir and not self._run_dir_inherited:
|
|
1670
|
+
self._write_run_manifest(
|
|
1671
|
+
run_id,
|
|
1672
|
+
status,
|
|
1673
|
+
started_at,
|
|
1674
|
+
arguments,
|
|
1675
|
+
resolved_seed,
|
|
1676
|
+
realized_name,
|
|
1677
|
+
annotations,
|
|
1678
|
+
)
|
|
1679
|
+
deactivate_output_root(output_root_token)
|
|
1680
|
+
run_context.exit_run()
|
|
1681
|
+
deactivate_context(context_token)
|
|
1682
|
+
|
|
1683
|
+
def _write_run_manifest(
|
|
1684
|
+
self,
|
|
1685
|
+
run_id,
|
|
1686
|
+
status,
|
|
1687
|
+
started_at,
|
|
1688
|
+
arguments,
|
|
1689
|
+
seed,
|
|
1690
|
+
realized_name=None,
|
|
1691
|
+
annotations=None,
|
|
1692
|
+
):
|
|
1693
|
+
"""Leave a record of the run beside the files it wrote.
|
|
1694
|
+
|
|
1695
|
+
A server run is in jobs.sqlite as well, but a CLI run has never been
|
|
1696
|
+
recorded anywhere, and a database on one machine cannot describe a
|
|
1697
|
+
directory copied to another. Paths are relative to the run directory
|
|
1698
|
+
so the directory keeps describing itself wherever it goes.
|
|
1699
|
+
"""
|
|
1700
|
+
from . import __version__
|
|
1701
|
+
|
|
1702
|
+
write_manifest(
|
|
1703
|
+
self._run_dir,
|
|
1704
|
+
{
|
|
1705
|
+
"run_id": run_id,
|
|
1706
|
+
# This run's ordinal among the workflow's runs - 'v4' in the
|
|
1707
|
+
# gallery. Recorded, never recomputed
|
|
1708
|
+
"version": self._run_version,
|
|
1709
|
+
"status": status,
|
|
1710
|
+
"started_at": started_at,
|
|
1711
|
+
# None on the manifest written as the run opens
|
|
1712
|
+
"finished_at": (
|
|
1713
|
+
None
|
|
1714
|
+
if status == "running"
|
|
1715
|
+
else datetime.now(timezone.utc).isoformat()
|
|
1716
|
+
),
|
|
1717
|
+
"dw_version": __version__,
|
|
1718
|
+
"device": str(get_device()),
|
|
1719
|
+
"workflow": {
|
|
1720
|
+
"id": self.name,
|
|
1721
|
+
"file": self.file_spec,
|
|
1722
|
+
"identity": workflow_identity(self.file_spec, self.name),
|
|
1723
|
+
# The realized copy beside this manifest, or null when
|
|
1724
|
+
# writing it did not land - the manifest is the only
|
|
1725
|
+
# place that difference is visible
|
|
1726
|
+
"realized": realized_name,
|
|
1727
|
+
# Annotations the schema has nowhere to put: which
|
|
1728
|
+
# stored prompts were inlined, and what each local
|
|
1729
|
+
# sub-workflow file held when it ran
|
|
1730
|
+
"prompts": (annotations or {}).get("prompts", []),
|
|
1731
|
+
"sub_workflows": (annotations or {}).get("sub_workflows", {}),
|
|
1732
|
+
},
|
|
1733
|
+
"seed": seed,
|
|
1734
|
+
"arguments": arguments or {},
|
|
1735
|
+
# What did not run, and why - a run says what it did not do
|
|
1736
|
+
# as well as what it did (#122)
|
|
1737
|
+
"elided_steps": self._elided_steps,
|
|
1738
|
+
"steps": [
|
|
1739
|
+
{
|
|
1740
|
+
**entry,
|
|
1741
|
+
"files": manifest_relative_files(
|
|
1742
|
+
entry.get("files"), self._run_dir
|
|
1743
|
+
),
|
|
1744
|
+
**_relative_shots(entry, self._run_dir),
|
|
1745
|
+
}
|
|
1746
|
+
for entry in self.manifest
|
|
1747
|
+
],
|
|
1748
|
+
},
|
|
1749
|
+
)
|
|
1750
|
+
|
|
1751
|
+
def _step_pipeline_key(self, step_name, cache_key):
|
|
1752
|
+
"""Record which cache key a step's pipeline lives under this run."""
|
|
1753
|
+
if not hasattr(self, "_pipeline_keys_by_step"):
|
|
1754
|
+
self._pipeline_keys_by_step = {}
|
|
1755
|
+
self._pipeline_keys_by_step[step_name] = cache_key
|
|
1756
|
+
|
|
1757
|
+
def create_step_action(
|
|
1758
|
+
self,
|
|
1759
|
+
step_definition,
|
|
1760
|
+
shared_components,
|
|
1761
|
+
previous_pipelines,
|
|
1762
|
+
default_seed,
|
|
1763
|
+
device,
|
|
1764
|
+
):
|
|
1765
|
+
"""
|
|
1766
|
+
Creates the appropriate action object based on step type:
|
|
1767
|
+
- Pipeline: Creates new pipeline or reuses cached one
|
|
1768
|
+
- Pipeline reference: References existing pipeline
|
|
1769
|
+
- Workflow: Loads and validates sub-workflow
|
|
1770
|
+
- Task: Creates task object
|
|
1771
|
+
"""
|
|
1772
|
+
# Handle pipeline creation
|
|
1773
|
+
if "pipeline" in step_definition:
|
|
1774
|
+
step_name = step_definition["name"]
|
|
1775
|
+
|
|
1776
|
+
# Pipelines are cached by what they load, not what step loads them
|
|
1777
|
+
cache_key = pipeline_cache_key(step_definition["pipeline"])
|
|
1778
|
+
self._step_pipeline_key(step_name, cache_key)
|
|
1779
|
+
get_context().touch_pipeline(cache_key)
|
|
1780
|
+
|
|
1781
|
+
# Check if pipeline already loaded in cache (GPU persistence)
|
|
1782
|
+
if cache_key in previous_pipelines:
|
|
1783
|
+
logger.debug(f"Reusing cached pipeline for step: {step_name}")
|
|
1784
|
+
cached_pipeline = previous_pipelines[cache_key]
|
|
1785
|
+
# The shared_components dict is fresh every run and only load()
|
|
1786
|
+
# fills it - a cache hit must republish or a later step's
|
|
1787
|
+
# reused_components finds nothing (impossible under the old
|
|
1788
|
+
# whole-file cache, the normal case under identity keys)
|
|
1789
|
+
cached_pipeline.publish_shared_components(shared_components)
|
|
1790
|
+
# Create new Pipeline wrapper with updated step definition
|
|
1791
|
+
# but reuse the loaded model from cache
|
|
1792
|
+
new_pipeline_wrapper = Pipeline(
|
|
1793
|
+
step_definition["pipeline"],
|
|
1794
|
+
default_seed,
|
|
1795
|
+
device,
|
|
1796
|
+
cached_pipeline.pipeline, # Reuse the actual loaded model
|
|
1797
|
+
output_dir=self.step_output_dir(step_definition),
|
|
1798
|
+
file_prefix=self.step_file_prefix(step_name),
|
|
1799
|
+
)
|
|
1800
|
+
# Set up generator with potentially new seed. no_generator is a
|
|
1801
|
+
# boolean - only an explicit true disables the generator - and the
|
|
1802
|
+
# generator lives on the pipeline's own device, which may override
|
|
1803
|
+
# the workflow default (the fresh-load path resolves it the same way)
|
|
1804
|
+
if not new_pipeline_wrapper.configuration.get("no_generator", False):
|
|
1805
|
+
logger.debug(
|
|
1806
|
+
"Setting up generator for cached pipeline with new arguments"
|
|
1807
|
+
)
|
|
1808
|
+
new_pipeline_wrapper.argument_template["generator"] = (
|
|
1809
|
+
torch.Generator(new_pipeline_wrapper.device).manual_seed(
|
|
1810
|
+
new_pipeline_wrapper.pipeline_definition.get(
|
|
1811
|
+
"seed", default_seed
|
|
1812
|
+
)
|
|
1813
|
+
)
|
|
1814
|
+
)
|
|
1815
|
+
|
|
1816
|
+
# A cache hit and a cold load look identical from the outside -
|
|
1817
|
+
# same step, same dot - and they differ by minutes
|
|
1818
|
+
emit_phase("cached", detail=new_pipeline_wrapper.name)
|
|
1819
|
+
return new_pipeline_wrapper
|
|
1820
|
+
|
|
1821
|
+
# Not in cache - a redefined step frees its previous model first,
|
|
1822
|
+
# so the swap never holds old and new stacks simultaneously
|
|
1823
|
+
prior_keys = getattr(self, "_prior_step_keys", {})
|
|
1824
|
+
prior_key = prior_keys.get(step_name)
|
|
1825
|
+
# Only this step's own variant: a key another step of THIS run
|
|
1826
|
+
# loads under NOW is that step's warm model, and releasing it
|
|
1827
|
+
# here would reload it cold a moment later while holding both
|
|
1828
|
+
# stacks. Judged on the other steps' current keys
|
|
1829
|
+
# (_running_pipeline_keys, recorded by run), not their prior
|
|
1830
|
+
# ones: when every step sharing one model variable changes at
|
|
1831
|
+
# once, each still has the old key as its prior key, and
|
|
1832
|
+
# nobody will load it again - holding it would be the two-stack
|
|
1833
|
+
# transition #150 fixed. _prior_step_keys is merged across every
|
|
1834
|
+
# job the worker has run, never pruned, so a name from an
|
|
1835
|
+
# earlier, unrelated workflow does not count either - it is not
|
|
1836
|
+
# among the running steps. If nothing touches the key this run,
|
|
1837
|
+
# the end-of-run sweep drops it.
|
|
1838
|
+
running_keys = getattr(self, "_running_pipeline_keys", {})
|
|
1839
|
+
still_shared = any(
|
|
1840
|
+
other != step_name and key == prior_key
|
|
1841
|
+
for other, key in running_keys.items()
|
|
1842
|
+
)
|
|
1843
|
+
if (
|
|
1844
|
+
prior_key
|
|
1845
|
+
and prior_key != cache_key
|
|
1846
|
+
and prior_key in previous_pipelines
|
|
1847
|
+
and not still_shared
|
|
1848
|
+
):
|
|
1849
|
+
logger.info(
|
|
1850
|
+
f"Step '{step_name}' was redefined - releasing its previous "
|
|
1851
|
+
"pipeline before loading the new one"
|
|
1852
|
+
)
|
|
1853
|
+
before = _allocated_mb()
|
|
1854
|
+
previous_pipelines.pop(prior_key, None)
|
|
1855
|
+
gc.collect()
|
|
1856
|
+
empty_device_cache()
|
|
1857
|
+
_release_host_caches(step_name)
|
|
1858
|
+
# Say so on the event stream, for the same reason the explicit
|
|
1859
|
+
# release does: without it a reload-on-top-of-a-resident-model
|
|
1860
|
+
# is indistinguishable from a cold load, and the difference is
|
|
1861
|
+
# whether the next thing that happens is an OOM kill (#150).
|
|
1862
|
+
# 'reason' separates it from the release a step asked for
|
|
1863
|
+
get_context().emit(
|
|
1864
|
+
"pipeline_released",
|
|
1865
|
+
step=step_name,
|
|
1866
|
+
reason="superseded",
|
|
1867
|
+
gpu_memory_allocated_mb=_allocated_mb(),
|
|
1868
|
+
gpu_memory_allocated_before_mb=before,
|
|
1869
|
+
)
|
|
1870
|
+
|
|
1871
|
+
logger.debug(f"Creating pipeline for step: {step_name}")
|
|
1872
|
+
pipeline = Pipeline(
|
|
1873
|
+
step_definition["pipeline"],
|
|
1874
|
+
default_seed,
|
|
1875
|
+
device,
|
|
1876
|
+
output_dir=self.step_output_dir(step_definition),
|
|
1877
|
+
file_prefix=self.step_file_prefix(step_name),
|
|
1878
|
+
)
|
|
1879
|
+
# Before the marker, not after it: a definition refused by the
|
|
1880
|
+
# trust gate must not have announced a load it never began, or a
|
|
1881
|
+
# consumer reading job events cannot tell 'refused before load'
|
|
1882
|
+
# from 'loaded, then refused' (#137)
|
|
1883
|
+
pipeline.check_trusted()
|
|
1884
|
+
# Loading is the longest silence in a run: weights, quantization,
|
|
1885
|
+
# adapters and placement all happen inside this call
|
|
1886
|
+
emit_phase("loading", detail=pipeline.name)
|
|
1887
|
+
pipeline.load(shared_components)
|
|
1888
|
+
previous_pipelines[cache_key] = pipeline
|
|
1889
|
+
return pipeline
|
|
1890
|
+
|
|
1891
|
+
# Handle pipeline reference
|
|
1892
|
+
if "pipeline_reference" in step_definition:
|
|
1893
|
+
logger.debug(
|
|
1894
|
+
f"Referencing existing pipeline for step: {step_definition['name']}"
|
|
1895
|
+
)
|
|
1896
|
+
pipeline_reference = step_definition["pipeline_reference"]
|
|
1897
|
+
reference_name = pipeline_reference["reference_name"]
|
|
1898
|
+
referenced_key = self._pipeline_keys_by_step.get(reference_name)
|
|
1899
|
+
if referenced_key is None or referenced_key not in previous_pipelines:
|
|
1900
|
+
raise ValueError(
|
|
1901
|
+
f"pipeline_reference '{reference_name}' does not name an "
|
|
1902
|
+
"earlier pipeline step in this run (or it was released)"
|
|
1903
|
+
)
|
|
1904
|
+
previous_pipeline = previous_pipelines[referenced_key]
|
|
1905
|
+
return Pipeline(
|
|
1906
|
+
pipeline_reference,
|
|
1907
|
+
default_seed,
|
|
1908
|
+
device,
|
|
1909
|
+
previous_pipeline.pipeline,
|
|
1910
|
+
output_dir=self.step_output_dir(step_definition),
|
|
1911
|
+
file_prefix=self.step_file_prefix(step_definition["name"]),
|
|
1912
|
+
)
|
|
1913
|
+
|
|
1914
|
+
# Handle sub-workflow
|
|
1915
|
+
if "workflow" in step_definition:
|
|
1916
|
+
logger.debug(f"Loading sub-workflow for step: {step_definition['name']}")
|
|
1917
|
+
workflow_reference = step_definition["workflow"]
|
|
1918
|
+
path = workflow_reference["path"]
|
|
1919
|
+
|
|
1920
|
+
try:
|
|
1921
|
+
# Sub-workflow steps are confined to the same directory this
|
|
1922
|
+
# workflow is (workflow_dir for a server-submitted run)
|
|
1923
|
+
confine_to = self.workflow_dir
|
|
1924
|
+
# Handle built-in workflows
|
|
1925
|
+
if path.startswith("builtin:"):
|
|
1926
|
+
builtin_name = path.replace("builtin:", "")
|
|
1927
|
+
# Validate builtin workflow name
|
|
1928
|
+
if (
|
|
1929
|
+
not builtin_name.endswith(".json")
|
|
1930
|
+
or "/" in builtin_name
|
|
1931
|
+
or "\\" in builtin_name
|
|
1932
|
+
):
|
|
1933
|
+
raise InvalidInputError(
|
|
1934
|
+
f"Invalid builtin workflow name: {builtin_name}. "
|
|
1935
|
+
"It must be a bare '<name>.json' filename with no "
|
|
1936
|
+
"path segments - 'builtin:' only looks in the "
|
|
1937
|
+
f"packaged workflows root: {builtin_root()}"
|
|
1938
|
+
)
|
|
1939
|
+
# Builtins ship inside the package, outside any
|
|
1940
|
+
# workflow_dir - confine them to their own directory
|
|
1941
|
+
# instead (the name check above already forbids escaping it)
|
|
1942
|
+
confine_to = os.path.join(
|
|
1943
|
+
os.path.dirname(os.path.abspath(__file__)), "workflows"
|
|
1944
|
+
)
|
|
1945
|
+
path = os.path.join(confine_to, builtin_name)
|
|
1946
|
+
# Everything else - a relative path, or a catalog name as
|
|
1947
|
+
# list_workflows reports it - goes through the search path.
|
|
1948
|
+
# A template under templates/ names a model config as
|
|
1949
|
+
# '../models/x.json', so a path relative to the referencing
|
|
1950
|
+
# file still resolves first and the '..' is collapsed here,
|
|
1951
|
+
# which is what lets the validator judge where the path
|
|
1952
|
+
# actually lands rather than refusing the spelling;
|
|
1953
|
+
# containment is still checked on the resolved path below.
|
|
1954
|
+
# An unconfined run (no workflow_dir - a bare CLI
|
|
1955
|
+
# invocation) used to rely on the '..' regex alone to stop a
|
|
1956
|
+
# relative reference from leaving the file's own directory;
|
|
1957
|
+
# normalizing the path removes that guard, so confine it to
|
|
1958
|
+
# the catalog root instead - the referencing file's nearest
|
|
1959
|
+
# ancestor literally named 'workflows', which still lets it
|
|
1960
|
+
# climb to a sibling folder like models/ but not out of the
|
|
1961
|
+
# catalog
|
|
1962
|
+
else:
|
|
1963
|
+
if confine_to is None and not os.path.isabs(path):
|
|
1964
|
+
confine_to = catalog_root_dir(self.file_spec)
|
|
1965
|
+
path, confine_to = resolve_sub_workflow(
|
|
1966
|
+
path, os.path.dirname(self.file_spec), confine_to
|
|
1967
|
+
)
|
|
1968
|
+
|
|
1969
|
+
# Validate the resolved path - confined when this workflow
|
|
1970
|
+
# itself is (an inline/server-submitted run), so a
|
|
1971
|
+
# sub-workflow step cannot escape that boundary
|
|
1972
|
+
validated_path = validate_workflow_path(path, confine_to)
|
|
1973
|
+
workflow = workflow_from_file(
|
|
1974
|
+
validated_path, self.output_dir, confine_to
|
|
1975
|
+
)
|
|
1976
|
+
|
|
1977
|
+
except SecurityError as e:
|
|
1978
|
+
logger.error(f"Security validation failed for sub-workflow {path}: {e}")
|
|
1979
|
+
raise
|
|
1980
|
+
|
|
1981
|
+
# this is where the arguments in the parent script are passed to the child workflow
|
|
1982
|
+
# they will already be populated with values from previous steps or parent variables
|
|
1983
|
+
workflow.workflow_definition["argument_template"] = workflow_reference.get(
|
|
1984
|
+
"arguments", {}
|
|
1985
|
+
)
|
|
1986
|
+
# A child left to itself draws its own random seed, which makes the
|
|
1987
|
+
# parent's seed stop short of the work it delegates. Inheriting it
|
|
1988
|
+
# keeps one seed reproducing the whole run; a child that names its
|
|
1989
|
+
# own still wins, the same way a step overrides its workflow
|
|
1990
|
+
workflow.workflow_definition.setdefault("seed", default_seed)
|
|
1991
|
+
# A seedless parent injects a seed that is fresh every run, so no
|
|
1992
|
+
# step of the child can ever hit - the child must not pay the
|
|
1993
|
+
# cache's deepcopy and Result pinning for it
|
|
1994
|
+
workflow._cache_enabled_by_parent = self._cache_enabled_this_run
|
|
1995
|
+
# One execution, one directory: the child writes into the
|
|
1996
|
+
# parent's run directory and leaves no manifest of its own - its
|
|
1997
|
+
# steps roll up into the parent's manifest already
|
|
1998
|
+
workflow._run_dir = self._run_dir
|
|
1999
|
+
workflow._run_dir_inherited = self._run_dir is not None
|
|
2000
|
+
workflow.validate()
|
|
2001
|
+
return workflow
|
|
2002
|
+
|
|
2003
|
+
logger.debug(f"Creating task for step: {step_definition['name']}")
|
|
2004
|
+
# Handle task creation
|
|
2005
|
+
task_definition = step_definition["task"]
|
|
2006
|
+
task = Task(task_definition, device, seed=default_seed)
|
|
2007
|
+
return task
|