diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/runs.py
ADDED
|
@@ -0,0 +1,768 @@
|
|
|
1
|
+
"""A run: the directory one execution of a workflow writes into, and the
|
|
2
|
+
manifest it leaves behind.
|
|
3
|
+
|
|
4
|
+
Output used to be laid out by where the workflow file sits - the subfolder
|
|
5
|
+
mirrored its position under the nearest directory literally named 'workflows',
|
|
6
|
+
so the *shape of a checkout* was the grouping key, and a workflow moved out of
|
|
7
|
+
that tree silently flattened. A run directory replaces that with the
|
|
8
|
+
workflow's own identity plus one directory per execution:
|
|
9
|
+
|
|
10
|
+
<output_dir>/<identity>/<run id>/
|
|
11
|
+
<the files the run wrote>
|
|
12
|
+
manifest.json
|
|
13
|
+
|
|
14
|
+
Everything one execution produced - intermediates, finals, and the record of
|
|
15
|
+
what made them - lands in one place, prunable and addressable as a unit, and
|
|
16
|
+
a rerun can no longer interleave its files with an earlier one's.
|
|
17
|
+
|
|
18
|
+
The old flat-ish layout stays available: DW_OUTPUT_LAYOUT=flat, an
|
|
19
|
+
'output_layout' setting of "flat", or --output-layout flat on dw.run and
|
|
20
|
+
dw.serve, for a caller whose scripts glob the output directory.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import contextvars
|
|
24
|
+
import hashlib
|
|
25
|
+
import json
|
|
26
|
+
import logging
|
|
27
|
+
import os
|
|
28
|
+
import re
|
|
29
|
+
from datetime import datetime, timezone
|
|
30
|
+
|
|
31
|
+
logger = logging.getLogger("dw")
|
|
32
|
+
|
|
33
|
+
RUN_LAYOUT = "run"
|
|
34
|
+
FLAT_LAYOUT = "flat"
|
|
35
|
+
LAYOUTS = (RUN_LAYOUT, FLAT_LAYOUT)
|
|
36
|
+
|
|
37
|
+
# Set by an entry point, and inherited by a spawned worker the way
|
|
38
|
+
# DW_PROMPT_DIR and DW_ASSET_DIR are
|
|
39
|
+
OUTPUT_LAYOUT_ENV_VAR = "DW_OUTPUT_LAYOUT"
|
|
40
|
+
|
|
41
|
+
MANIFEST_FILE_NAME = "manifest.json"
|
|
42
|
+
|
|
43
|
+
# The workflow a run actually ran, written beside its manifest. Named
|
|
44
|
+
# 'workflow.json' rather than something run-specific because the directory
|
|
45
|
+
# already says which run it is, and 'python -m dw.run workflow.json' from
|
|
46
|
+
# inside it is the whole reproduction story
|
|
47
|
+
REALIZED_FILE_NAME = "workflow.json"
|
|
48
|
+
|
|
49
|
+
# The prefix marking a value as a reference to a file an earlier run wrote.
|
|
50
|
+
# Like 'asset:', it stands for a path - what a previous run made is an input
|
|
51
|
+
# like any other, and multi-stage work is what a workflow engine is for
|
|
52
|
+
OUTPUT_PREFIX = "output:"
|
|
53
|
+
|
|
54
|
+
# The segment that means "the newest run of this workflow that has the
|
|
55
|
+
# file", so a workflow can name the stage before it without being edited
|
|
56
|
+
# after every run - see _resolve_segments for why it is not simply the
|
|
57
|
+
# newest run directory
|
|
58
|
+
LATEST = "latest"
|
|
59
|
+
|
|
60
|
+
# 'v4' in the run-id position of an 'output:' reference: the run whose
|
|
61
|
+
# ordinal is 4 - the number the gallery shows and an agent quotes
|
|
62
|
+
_VERSION_SELECTOR = re.compile(r"^v([1-9][0-9]*)$")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def version_selector(segment):
|
|
66
|
+
"""The ordinal a 'v<N>' segment names, or None for any other segment."""
|
|
67
|
+
match = _VERSION_SELECTOR.match(segment)
|
|
68
|
+
return int(match.group(1)) if match else None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# What a run id looks like: a UTC timestamp and a short digest of the spec.
|
|
72
|
+
# The pattern is not only documentation - the gallery reads it to group a
|
|
73
|
+
# workflow's runs under one folder rather than listing every run separately
|
|
74
|
+
# The trailing counter appears only when two runs of the same spec start in
|
|
75
|
+
# the same second - see run_directory
|
|
76
|
+
RUN_ID_PATTERN = re.compile(r"^\d{8}-\d{6}-[0-9a-f]{8}(-\d+)?$")
|
|
77
|
+
|
|
78
|
+
# Characters allowed in a path segment derived from a workflow's name or file
|
|
79
|
+
_UNSAFE_SEGMENT_CHARACTERS = re.compile(r"[^A-Za-z0-9_.-]+")
|
|
80
|
+
|
|
81
|
+
# The synthetic file name workflow_from_definition gives an inline workflow -
|
|
82
|
+
# it carries a directory, not an identity
|
|
83
|
+
INLINE_FILE_NAME = "__inline__"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# The output root of the run in progress, so an 'output:' reference resolves
|
|
87
|
+
# against the directory this run was told to write to rather than guessing
|
|
88
|
+
# one. Set by Workflow.run; a reference realized outside any run falls back
|
|
89
|
+
# to the workspace's outputs
|
|
90
|
+
_active_output_root = contextvars.ContextVar("dw_output_root", default=None)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def activate_output_root(root):
|
|
94
|
+
"""Make an output root the active one; returns a token for deactivate."""
|
|
95
|
+
return _active_output_root.set(root)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def deactivate_output_root(token):
|
|
99
|
+
_active_output_root.reset(token)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def output_root():
|
|
103
|
+
"""The output directory 'output:' references resolve against."""
|
|
104
|
+
active = _active_output_root.get()
|
|
105
|
+
if active:
|
|
106
|
+
return active
|
|
107
|
+
|
|
108
|
+
from .workspace import resolve_workspace
|
|
109
|
+
|
|
110
|
+
return resolve_workspace().outputs
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def is_output_reference(value):
|
|
114
|
+
"""Whether a value references a file an earlier run wrote."""
|
|
115
|
+
return isinstance(value, str) and value.startswith(OUTPUT_PREFIX)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _runs_newest_first(directory):
|
|
119
|
+
"""The run directories inside a workflow's output folder, newest first.
|
|
120
|
+
|
|
121
|
+
Run ids start with a UTC timestamp, so sort order is age order - no stat
|
|
122
|
+
calls, and no dependence on mtimes that a copy would have rewritten
|
|
123
|
+
anyway. Empty when the directory holds no runs, or is not there.
|
|
124
|
+
"""
|
|
125
|
+
try:
|
|
126
|
+
return sorted(
|
|
127
|
+
(
|
|
128
|
+
name
|
|
129
|
+
for name in os.listdir(directory)
|
|
130
|
+
if is_run_id(name) and os.path.isdir(os.path.join(directory, name))
|
|
131
|
+
),
|
|
132
|
+
key=run_id_sort_key,
|
|
133
|
+
reverse=True,
|
|
134
|
+
)
|
|
135
|
+
except OSError:
|
|
136
|
+
return []
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _resolve_segments(directory, parts, reference, root):
|
|
140
|
+
"""Build the path a name stands for, expanding 'latest' or 'v<N>' where
|
|
141
|
+
it names a run.
|
|
142
|
+
|
|
143
|
+
'latest' means the newest run *that has the file*, not the newest run
|
|
144
|
+
directory: a run that failed part way, or one whose every step was a
|
|
145
|
+
cache hit, leaves a directory holding only its manifest, and a
|
|
146
|
+
second-stage workflow pointed at that would find nothing where the
|
|
147
|
+
stage before it plainly produced something. So the runs are tried
|
|
148
|
+
newest first and the first one holding the rest of the name wins.
|
|
149
|
+
|
|
150
|
+
'v<N>' means the run whose recorded ordinal is N - the 'v4' the gallery
|
|
151
|
+
shows - so the number quoted to a person is also a name a workflow can
|
|
152
|
+
take. Unlike 'latest' it picks exactly one run: a v4 that did not write
|
|
153
|
+
the file is an error, not a reason to try v3.
|
|
154
|
+
|
|
155
|
+
Only a segment standing where run directories are is a run selector. A
|
|
156
|
+
'latest' or 'v4' segment in a directory that holds no runs is a name
|
|
157
|
+
like any other, so a workflow or a file called either stays reachable.
|
|
158
|
+
|
|
159
|
+
Returns the path, or None when runs were found and none of them holds
|
|
160
|
+
the file.
|
|
161
|
+
"""
|
|
162
|
+
if not parts:
|
|
163
|
+
return directory
|
|
164
|
+
part, rest = parts[0], parts[1:]
|
|
165
|
+
if part == LATEST:
|
|
166
|
+
runs = _runs_newest_first(directory)
|
|
167
|
+
if runs:
|
|
168
|
+
for run in runs:
|
|
169
|
+
candidate = _resolve_segments(
|
|
170
|
+
os.path.join(directory, run), rest, reference, root
|
|
171
|
+
)
|
|
172
|
+
if candidate and os.path.isfile(candidate):
|
|
173
|
+
return candidate
|
|
174
|
+
return None
|
|
175
|
+
if not os.path.exists(os.path.join(directory, part)):
|
|
176
|
+
raise ValueError(
|
|
177
|
+
f"No runs yet under {os.path.relpath(directory, root)} - "
|
|
178
|
+
f"'{reference}' names the newest run of a workflow that "
|
|
179
|
+
f"has not produced one"
|
|
180
|
+
)
|
|
181
|
+
wanted = version_selector(part)
|
|
182
|
+
if wanted is not None and _runs_newest_first(directory):
|
|
183
|
+
versions = run_versions(directory)
|
|
184
|
+
matching = [run for run, version in versions.items() if version == wanted]
|
|
185
|
+
if not matching:
|
|
186
|
+
held = ", ".join(f"v{v}" for v in sorted(set(versions.values())))
|
|
187
|
+
raise ValueError(
|
|
188
|
+
f"No run v{wanted} under {os.path.relpath(directory, root)} - "
|
|
189
|
+
f"'{reference}' names a run by its version, and the runs "
|
|
190
|
+
f"there are {held}"
|
|
191
|
+
)
|
|
192
|
+
# Normally one; two only where history predating versions could
|
|
193
|
+
# not be ranked beneath the first recorded number. Newest first,
|
|
194
|
+
# as 'latest' would try them
|
|
195
|
+
for run in sorted(matching, key=run_id_sort_key, reverse=True):
|
|
196
|
+
candidate = _resolve_segments(
|
|
197
|
+
os.path.join(directory, run), rest, reference, root
|
|
198
|
+
)
|
|
199
|
+
if candidate and os.path.isfile(candidate):
|
|
200
|
+
return candidate
|
|
201
|
+
return None
|
|
202
|
+
return _resolve_segments(os.path.join(directory, part), rest, reference, root)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def resolve_output_reference(reference, root=None):
|
|
206
|
+
"""Resolve an 'output:' reference to the file it names.
|
|
207
|
+
|
|
208
|
+
The name is a path under the output directory - '<workflow>/<run
|
|
209
|
+
id>/<file>' - and the run id may be written as 'latest', which resolves
|
|
210
|
+
to the newest run of that workflow that holds the file, or as 'v<N>',
|
|
211
|
+
the run whose version is N. 'latest' is what
|
|
212
|
+
lets a second-stage workflow name the first stage's product without
|
|
213
|
+
being edited after every run, and without breaking when the newest run
|
|
214
|
+
failed or reused cached files and so wrote none of its own.
|
|
215
|
+
|
|
216
|
+
Args:
|
|
217
|
+
reference: The 'output:...' string
|
|
218
|
+
root: The output directory to resolve against; defaults to the run
|
|
219
|
+
in progress, else the workspace's outputs
|
|
220
|
+
|
|
221
|
+
Returns:
|
|
222
|
+
The validated absolute path of the file
|
|
223
|
+
|
|
224
|
+
Raises:
|
|
225
|
+
InvalidInputError: If the name is not a valid output name
|
|
226
|
+
PathTraversalError: If the name escapes the output directory
|
|
227
|
+
ValueError: If no such run or file exists
|
|
228
|
+
"""
|
|
229
|
+
from .security import validate_output_reference, validate_path
|
|
230
|
+
|
|
231
|
+
name = validate_output_reference(reference.removeprefix(OUTPUT_PREFIX).strip())
|
|
232
|
+
root = root or output_root()
|
|
233
|
+
|
|
234
|
+
resolved = _resolve_segments(root, name.split("/"), reference, root)
|
|
235
|
+
if resolved is None:
|
|
236
|
+
raise ValueError(
|
|
237
|
+
f"Output '{name}' not found under {root} - no run of that workflow "
|
|
238
|
+
f"holds the file. A run that failed, or reused every step from the "
|
|
239
|
+
f"cache, leaves only its manifest behind"
|
|
240
|
+
)
|
|
241
|
+
# Containment is checked once, on the whole path, after 'latest' has
|
|
242
|
+
# been expanded - so what is validated is the real directory it named.
|
|
243
|
+
# The segments themselves were pattern-checked, so this guards symlinks
|
|
244
|
+
resolved = validate_path(resolved, root)
|
|
245
|
+
|
|
246
|
+
if not os.path.isfile(resolved):
|
|
247
|
+
raise ValueError(
|
|
248
|
+
f"Output '{name}' not found under {root} - an 'output:' reference "
|
|
249
|
+
f"names a file an earlier run wrote, like "
|
|
250
|
+
f"'output:ltx2/Gyre/latest/Gyre-still.0-0.0.png'"
|
|
251
|
+
)
|
|
252
|
+
logger.debug(f"Resolved {reference} to {resolved}")
|
|
253
|
+
return resolved
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def fetch_output(reference, root=None):
|
|
257
|
+
"""The path an 'output:' reference names, for whatever loads paths."""
|
|
258
|
+
return resolve_output_reference(reference, root)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def output_layout():
|
|
262
|
+
"""Whether runs get their own directory ('run') or write into the output
|
|
263
|
+
directory the way they did before ('flat').
|
|
264
|
+
|
|
265
|
+
Read at call time so a worker subprocess and a test see the current
|
|
266
|
+
value, the same as every other directory question.
|
|
267
|
+
"""
|
|
268
|
+
from_environment = os.environ.get(OUTPUT_LAYOUT_ENV_VAR)
|
|
269
|
+
if from_environment in LAYOUTS:
|
|
270
|
+
return from_environment
|
|
271
|
+
|
|
272
|
+
from .settings import load_settings
|
|
273
|
+
|
|
274
|
+
from_settings = load_settings().output_layout
|
|
275
|
+
return from_settings if from_settings in LAYOUTS else RUN_LAYOUT
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def set_output_layout(layout):
|
|
279
|
+
"""Pin the layout for this process and anything it spawns."""
|
|
280
|
+
if layout not in LAYOUTS:
|
|
281
|
+
raise ValueError(f"Unknown output layout: {layout}")
|
|
282
|
+
os.environ[OUTPUT_LAYOUT_ENV_VAR] = layout
|
|
283
|
+
return layout
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _safe_segment(text):
|
|
287
|
+
cleaned = _UNSAFE_SEGMENT_CHARACTERS.sub("_", str(text)).strip("._")
|
|
288
|
+
return cleaned or "workflow"
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def workflow_identity(file_spec, workflow_id=None):
|
|
292
|
+
"""What names this workflow's outputs, as a relative path.
|
|
293
|
+
|
|
294
|
+
A workflow's position under a 'workflows' tree still reads as its
|
|
295
|
+
identity when it has one - 'workflows/ltx2/Gyre.json' is 'ltx2/Gyre' -
|
|
296
|
+
because that is the organization a user already chose. Outside such a
|
|
297
|
+
tree the file's own name is the identity, and an inline definition,
|
|
298
|
+
which has no file, is named by its workflow id.
|
|
299
|
+
|
|
300
|
+
The result is always a relative path of safe segments: it is joined onto
|
|
301
|
+
the output directory, and nothing about it is allowed to leave.
|
|
302
|
+
"""
|
|
303
|
+
name = None
|
|
304
|
+
subfolder = ""
|
|
305
|
+
if file_spec:
|
|
306
|
+
base = os.path.basename(file_spec)
|
|
307
|
+
stem = os.path.splitext(base)[0]
|
|
308
|
+
if stem and stem != INLINE_FILE_NAME:
|
|
309
|
+
name = stem
|
|
310
|
+
directory = os.path.dirname(os.path.abspath(file_spec))
|
|
311
|
+
parts = os.path.normpath(directory).split(os.sep)
|
|
312
|
+
try:
|
|
313
|
+
# The last 'workflows' segment wins, matching the packaged
|
|
314
|
+
# dw/workflows tree when a checkout has a top-level one too
|
|
315
|
+
index = len(parts) - 1 - parts[::-1].index("workflows")
|
|
316
|
+
except ValueError:
|
|
317
|
+
index = None
|
|
318
|
+
if index is not None and index + 1 < len(parts):
|
|
319
|
+
subfolder = os.path.join(*(_safe_segment(p) for p in parts[index + 1 :]))
|
|
320
|
+
|
|
321
|
+
name = _safe_segment(name or workflow_id or "workflow")
|
|
322
|
+
return os.path.join(subfolder, name) if subfolder else name
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def new_run_id(spec=None, now=None):
|
|
326
|
+
"""An identifier for one execution: a UTC timestamp, then eight hex
|
|
327
|
+
digits of the spec that produced it.
|
|
328
|
+
|
|
329
|
+
The timestamp is what sorts and what a person reads; the digest is what
|
|
330
|
+
tells two runs of the same second apart and makes a rerun of an edited
|
|
331
|
+
workflow visibly different from a rerun of the same one. A server job
|
|
332
|
+
could have used its job id, but a CLI run has none, and one scheme
|
|
333
|
+
everywhere is what lets anything reading the directory tree - the
|
|
334
|
+
gallery, a future history rebuild - understand both.
|
|
335
|
+
"""
|
|
336
|
+
stamp = (now or datetime.now(timezone.utc)).strftime("%Y%m%d-%H%M%S")
|
|
337
|
+
try:
|
|
338
|
+
material = json.dumps(spec, sort_keys=True, default=str)
|
|
339
|
+
except (TypeError, ValueError):
|
|
340
|
+
material = repr(spec)
|
|
341
|
+
digest = hashlib.sha256(material.encode("utf-8", "replace")).hexdigest()[:8]
|
|
342
|
+
return f"{stamp}-{digest}"
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def is_run_id(segment):
|
|
346
|
+
"""Whether a path segment is a run id this module generated."""
|
|
347
|
+
return bool(RUN_ID_PATTERN.match(segment or ""))
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def split_run_path(relative_path):
|
|
351
|
+
"""The three parts of a run-relative path: (identity, run id, subfolder).
|
|
352
|
+
|
|
353
|
+
'ltx2/Gyre/20260905-181530-a1b2c3d4/final/still.png' ->
|
|
354
|
+
('ltx2/Gyre', '20260905-181530-a1b2c3d4', 'final'). The run id is
|
|
355
|
+
found wherever it sits, not only as the last directory - a step's
|
|
356
|
+
'subfolder' puts segments after it. A path with no run id in it (the
|
|
357
|
+
flat layout) has its whole directory as identity and nothing else,
|
|
358
|
+
which is what it was before subfolders existed.
|
|
359
|
+
|
|
360
|
+
Only the first segment matching RUN_ID_PATTERN counts. A workflow
|
|
361
|
+
*file* named in that shape would produce a matching identity segment;
|
|
362
|
+
that is unsupported rather than impossible.
|
|
363
|
+
"""
|
|
364
|
+
parts = [part for part in (relative_path or "").split("/") if part]
|
|
365
|
+
directory = parts[:-1]
|
|
366
|
+
for position, segment in enumerate(directory):
|
|
367
|
+
if is_run_id(segment):
|
|
368
|
+
return (
|
|
369
|
+
"/".join(directory[:position]),
|
|
370
|
+
segment,
|
|
371
|
+
"/".join(directory[position + 1 :]),
|
|
372
|
+
)
|
|
373
|
+
return "/".join(directory), "", ""
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def strip_run_id(relative_path):
|
|
377
|
+
"""The workflow identity a run-relative path belongs to.
|
|
378
|
+
|
|
379
|
+
'ltx2/Gyre/20260905-181530-a1b2c3d4/still-0.png' -> 'ltx2/Gyre', and
|
|
380
|
+
the same with a subfolder after the run id. A path with no run id in it
|
|
381
|
+
comes back with its own directory unchanged, which is what a
|
|
382
|
+
flat-layout output does.
|
|
383
|
+
"""
|
|
384
|
+
return split_run_path(relative_path)[0]
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
# The key a run's ordinal is recorded under in its manifest. It is assigned
|
|
388
|
+
# once, when the run directory is opened, and never recomputed - which is the
|
|
389
|
+
# whole point: a number quoted in conversation has to still mean the same run
|
|
390
|
+
# after a sibling is deleted. Deleting a middle run leaves a gap, and so does
|
|
391
|
+
# a run that wrote no media (it failed, or every step was reused from the
|
|
392
|
+
# cache): it took a number and has nothing in the gallery to show under it
|
|
393
|
+
RUN_VERSION_KEY = "version"
|
|
394
|
+
|
|
395
|
+
# The length of a run id before any '-N' counter a same-second rerun takes
|
|
396
|
+
_RUN_ID_BASE_LENGTH = len("20260101-000000-00000000")
|
|
397
|
+
|
|
398
|
+
# Recorded ordinals by manifest path, keyed on the manifest's stat so an
|
|
399
|
+
# edited or replaced manifest is read again. A recorded number never changes,
|
|
400
|
+
# so this is what keeps a gallery listing from parsing every manifest under
|
|
401
|
+
# the output root on every call
|
|
402
|
+
_recorded_versions = {}
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def run_id_sort_key(run_id):
|
|
406
|
+
"""Order run ids oldest first, with a rerun's '-N' counter compared as a
|
|
407
|
+
number - lexically '-10' would sort before '-2'."""
|
|
408
|
+
base, counter = run_id[:_RUN_ID_BASE_LENGTH], run_id[_RUN_ID_BASE_LENGTH + 1 :]
|
|
409
|
+
return (base, int(counter) if counter.isdigit() else 1)
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _read_manifest(run_dir):
|
|
413
|
+
"""A run's manifest as a dict, or None when it is missing or unreadable."""
|
|
414
|
+
try:
|
|
415
|
+
with open(os.path.join(run_dir, MANIFEST_FILE_NAME)) as file:
|
|
416
|
+
manifest = json.load(file)
|
|
417
|
+
except (OSError, ValueError):
|
|
418
|
+
return None
|
|
419
|
+
return manifest if isinstance(manifest, dict) else None
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def _valid_version(version):
|
|
423
|
+
# bool is an int subclass, and True is not version 1
|
|
424
|
+
if isinstance(version, bool) or not isinstance(version, int):
|
|
425
|
+
return None
|
|
426
|
+
return version if version > 0 else None
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def _recorded_version(run_dir):
|
|
430
|
+
"""The ordinal a run recorded for itself, or None.
|
|
431
|
+
|
|
432
|
+
None covers every way the number can be missing: a run made before this
|
|
433
|
+
field existed, one killed before its manifest landed, and one whose
|
|
434
|
+
manifest cannot be parsed. All three are ranked rather than trusted.
|
|
435
|
+
"""
|
|
436
|
+
path = os.path.join(run_dir, MANIFEST_FILE_NAME)
|
|
437
|
+
try:
|
|
438
|
+
stat = os.stat(path)
|
|
439
|
+
except OSError:
|
|
440
|
+
_recorded_versions.pop(path, None)
|
|
441
|
+
return None
|
|
442
|
+
signature = (stat.st_mtime_ns, stat.st_size, stat.st_ino)
|
|
443
|
+
cached = _recorded_versions.get(path)
|
|
444
|
+
if cached is not None and cached[0] == signature:
|
|
445
|
+
return cached[1]
|
|
446
|
+
manifest = _read_manifest(run_dir)
|
|
447
|
+
version = _valid_version(manifest.get(RUN_VERSION_KEY)) if manifest else None
|
|
448
|
+
_recorded_versions[path] = (signature, version)
|
|
449
|
+
return version
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def _run_ids(identity_dir):
|
|
453
|
+
"""Every run directory under one workflow identity, oldest first.
|
|
454
|
+
|
|
455
|
+
Run ids sort by their UTC timestamp, so this order is chronological
|
|
456
|
+
to the second - the same property `latest` relies on. Within one second
|
|
457
|
+
the spec digest decides, which is arbitrary but stable; nothing here
|
|
458
|
+
needs finer ordering than that.
|
|
459
|
+
"""
|
|
460
|
+
try:
|
|
461
|
+
entries = os.listdir(identity_dir)
|
|
462
|
+
except OSError:
|
|
463
|
+
return []
|
|
464
|
+
return sorted(
|
|
465
|
+
(
|
|
466
|
+
name
|
|
467
|
+
for name in entries
|
|
468
|
+
if is_run_id(name) and os.path.isdir(os.path.join(identity_dir, name))
|
|
469
|
+
),
|
|
470
|
+
key=run_id_sort_key,
|
|
471
|
+
)
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def _ranked_versions(identity_dir):
|
|
475
|
+
"""({run id: version}, {run id: recorded version or None})."""
|
|
476
|
+
run_ids = _run_ids(identity_dir)
|
|
477
|
+
recorded = {
|
|
478
|
+
run_id: _recorded_version(os.path.join(identity_dir, run_id))
|
|
479
|
+
for run_id in run_ids
|
|
480
|
+
}
|
|
481
|
+
# Room beneath the lowest recorded number for the unrecorded runs that
|
|
482
|
+
# come before it. Where there is not enough room the sequence starts at
|
|
483
|
+
# 1 and the recorded numbers stand: a duplicate is better than
|
|
484
|
+
# renumbering a run someone has already been told the number of
|
|
485
|
+
leading = 0
|
|
486
|
+
for run_id in run_ids:
|
|
487
|
+
if recorded[run_id] is not None:
|
|
488
|
+
break
|
|
489
|
+
leading += 1
|
|
490
|
+
next_number = 1
|
|
491
|
+
if leading < len(run_ids):
|
|
492
|
+
next_number = max(1, recorded[run_ids[leading]] - leading)
|
|
493
|
+
versions = {}
|
|
494
|
+
for run_id in run_ids:
|
|
495
|
+
if recorded[run_id] is not None:
|
|
496
|
+
versions[run_id] = recorded[run_id]
|
|
497
|
+
# Never backwards: two runs of one second can sort in the
|
|
498
|
+
# opposite order to their numbers, and an unrecorded run after
|
|
499
|
+
# them must not take a number the higher one already holds
|
|
500
|
+
next_number = max(next_number, recorded[run_id] + 1)
|
|
501
|
+
else:
|
|
502
|
+
versions[run_id] = next_number
|
|
503
|
+
next_number += 1
|
|
504
|
+
return versions, recorded
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def run_versions(identity_dir):
|
|
508
|
+
"""Every run of one workflow mapped to its ordinal: {run id: version}.
|
|
509
|
+
|
|
510
|
+
A run that recorded a version keeps it verbatim - that is what makes the
|
|
511
|
+
number survive a sibling being deleted. A run that recorded none (made
|
|
512
|
+
before the field existed, or killed before its manifest landed) is
|
|
513
|
+
ranked into the sequence around it: the unrecorded runs *older* than
|
|
514
|
+
every recorded one take the numbers just beneath the lowest recorded
|
|
515
|
+
one, so history that predates the field lands where it belongs, and an
|
|
516
|
+
unrecorded run anywhere later continues from the highest number before
|
|
517
|
+
it. Ordering is by run id, which is chronological.
|
|
518
|
+
|
|
519
|
+
Read only. A ranked number is only as stable as its neighbours until
|
|
520
|
+
`record_run_versions` writes it down.
|
|
521
|
+
"""
|
|
522
|
+
return _ranked_versions(identity_dir)[0]
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def record_run_versions(identity_dir):
|
|
526
|
+
"""Write each ranked number into the manifest of a run that has one but
|
|
527
|
+
records no version, and return every run's ordinal.
|
|
528
|
+
|
|
529
|
+
A ranked number moves when an older unrecorded sibling is deleted, so
|
|
530
|
+
runs made before the field existed are pinned the first time anything
|
|
531
|
+
writes under their workflow: a new run opening, or a run directory
|
|
532
|
+
being deleted. The listing never writes. A run with no manifest at all
|
|
533
|
+
is left alone - writing one would invent a record of a run nobody
|
|
534
|
+
recorded - and stays ranked.
|
|
535
|
+
|
|
536
|
+
Best effort: a manifest that cannot be rewritten keeps its ranked number.
|
|
537
|
+
"""
|
|
538
|
+
versions, recorded = _ranked_versions(identity_dir)
|
|
539
|
+
for run_id, version in versions.items():
|
|
540
|
+
if recorded[run_id] is not None:
|
|
541
|
+
continue
|
|
542
|
+
run_dir = os.path.join(identity_dir, run_id)
|
|
543
|
+
manifest = _read_manifest(run_dir)
|
|
544
|
+
if manifest is None:
|
|
545
|
+
continue
|
|
546
|
+
manifest[RUN_VERSION_KEY] = version
|
|
547
|
+
path = os.path.join(run_dir, MANIFEST_FILE_NAME)
|
|
548
|
+
partial = f"{path}.partial"
|
|
549
|
+
try:
|
|
550
|
+
with open(partial, "w") as file:
|
|
551
|
+
json.dump(manifest, file, indent=2, default=str)
|
|
552
|
+
os.replace(partial, path)
|
|
553
|
+
except OSError as e:
|
|
554
|
+
logger.warning(f"Could not record version {version} in {path}: {e}")
|
|
555
|
+
return versions
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def assign_run_version(output_dir, identity):
|
|
559
|
+
"""The ordinal the run about to open under `identity` takes.
|
|
560
|
+
|
|
561
|
+
One past the highest ordinal any sibling holds - not one past the newest
|
|
562
|
+
run's, because run ids are chronological only across seconds: two runs
|
|
563
|
+
started in the same second are ordered by their spec digest, so the last
|
|
564
|
+
id is not reliably the highest number. Three quick reruns are exactly
|
|
565
|
+
that case.
|
|
566
|
+
|
|
567
|
+
Pins the ranked numbers of older runs on the way (`record_run_versions`),
|
|
568
|
+
so history that predates the field stops moving once a new run joins it.
|
|
569
|
+
Sharing that ranking rather than deriving the maximum separately is what
|
|
570
|
+
keeps the number assigned here and the number the gallery reports from
|
|
571
|
+
drifting apart.
|
|
572
|
+
|
|
573
|
+
Best effort, like everything else that writes a run's bookkeeping: a
|
|
574
|
+
directory that cannot be read yields 1 rather than failing the run.
|
|
575
|
+
"""
|
|
576
|
+
versions = record_run_versions(os.path.join(output_dir, identity))
|
|
577
|
+
return max(versions.values(), default=0) + 1
|
|
578
|
+
|
|
579
|
+
|
|
580
|
+
def run_directory(output_dir, file_spec, workflow_id, run_id):
|
|
581
|
+
"""Where one execution writes: <output_dir>/<identity>/<run id>.
|
|
582
|
+
|
|
583
|
+
One execution gets one directory, so a run id already taken - two runs
|
|
584
|
+
of the same spec started in the same second, which is what a quick
|
|
585
|
+
rerun is - takes a counter rather than writing into the earlier run's
|
|
586
|
+
directory and burying its manifest.
|
|
587
|
+
"""
|
|
588
|
+
base = os.path.join(output_dir, workflow_identity(file_spec, workflow_id), run_id)
|
|
589
|
+
candidate = base
|
|
590
|
+
counter = 1
|
|
591
|
+
while os.path.exists(candidate):
|
|
592
|
+
counter += 1
|
|
593
|
+
candidate = f"{base}-{counter}"
|
|
594
|
+
return candidate
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
def write_manifest(run_dir, manifest):
|
|
598
|
+
"""Record what a run did, beside what it made.
|
|
599
|
+
|
|
600
|
+
A server run is already in jobs.sqlite, but a CLI run has never been
|
|
601
|
+
recorded anywhere, and history that lives only in a database cannot
|
|
602
|
+
survive the directory being moved to another machine. Never fatal: a
|
|
603
|
+
run that produced its files has succeeded whether or not this lands.
|
|
604
|
+
"""
|
|
605
|
+
path = os.path.join(run_dir, MANIFEST_FILE_NAME)
|
|
606
|
+
try:
|
|
607
|
+
os.makedirs(run_dir, exist_ok=True)
|
|
608
|
+
with open(path, "w") as file:
|
|
609
|
+
json.dump(manifest, file, indent=2, default=str)
|
|
610
|
+
except OSError as e:
|
|
611
|
+
logger.warning(f"Could not write {path}: {e}")
|
|
612
|
+
return None
|
|
613
|
+
return path
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
def write_realized_workflow(run_dir, realized):
|
|
617
|
+
"""Leave the workflow that produced a run beside what it produced.
|
|
618
|
+
|
|
619
|
+
The submitted definition says what was asked for; this says what ran -
|
|
620
|
+
arguments folded in, the seed pinned, stored prompts inlined,
|
|
621
|
+
'output:latest' resolved. Best effort, exactly like write_manifest: a run
|
|
622
|
+
that produced its files has succeeded whether or not this lands.
|
|
623
|
+
"""
|
|
624
|
+
path = os.path.join(run_dir, REALIZED_FILE_NAME)
|
|
625
|
+
try:
|
|
626
|
+
os.makedirs(run_dir, exist_ok=True)
|
|
627
|
+
with open(path, "w") as file:
|
|
628
|
+
json.dump(realized, file, indent=2, default=str)
|
|
629
|
+
except (OSError, TypeError, ValueError) as e:
|
|
630
|
+
logger.warning(f"Could not write {path}: {e}")
|
|
631
|
+
return None
|
|
632
|
+
return path
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
def manifest_relative_files(files, run_dir):
|
|
636
|
+
"""A run's file paths as the manifest records them: relative to the run
|
|
637
|
+
directory, so the directory can be moved or copied and still describe
|
|
638
|
+
itself. A file from an earlier run - what a step cache hit republishes -
|
|
639
|
+
is outside this directory and stays absolute.
|
|
640
|
+
"""
|
|
641
|
+
recorded = []
|
|
642
|
+
for path in files or []:
|
|
643
|
+
try:
|
|
644
|
+
relative = os.path.relpath(path, run_dir)
|
|
645
|
+
except ValueError: # different drive on Windows
|
|
646
|
+
recorded.append(path)
|
|
647
|
+
continue
|
|
648
|
+
recorded.append(
|
|
649
|
+
path if relative.startswith(os.pardir) else relative.replace(os.sep, "/")
|
|
650
|
+
)
|
|
651
|
+
return recorded
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def recorded_shots(output_root, relative_path):
|
|
655
|
+
"""The shot boundaries the run's manifest records for one of its files.
|
|
656
|
+
|
|
657
|
+
None when the file is not in a run directory (the flat layout), its run
|
|
658
|
+
has no readable manifest, or no step recorded shots for it - a file that
|
|
659
|
+
was not joined from shots, or one written before shots were recorded.
|
|
660
|
+
"""
|
|
661
|
+
from .security import SecurityError, validate_path
|
|
662
|
+
from .shots import shots_for_file
|
|
663
|
+
|
|
664
|
+
folder, run_id, _subfolder = split_run_path(relative_path)
|
|
665
|
+
if not run_id:
|
|
666
|
+
return None
|
|
667
|
+
# Every route resolves the file itself through validate_path first, so
|
|
668
|
+
# this is contained already; checking the run directory too keeps that
|
|
669
|
+
# true for a caller that doesn't, and lets static analysis see it
|
|
670
|
+
try:
|
|
671
|
+
run_dir = validate_path(os.path.join(output_root, folder, run_id), output_root)
|
|
672
|
+
except SecurityError:
|
|
673
|
+
return None
|
|
674
|
+
manifest = _read_manifest(run_dir)
|
|
675
|
+
if manifest is None:
|
|
676
|
+
return None
|
|
677
|
+
prefix = f"{folder}/{run_id}/" if folder else f"{run_id}/"
|
|
678
|
+
own = relative_path[len(prefix) :] if relative_path.startswith(prefix) else None
|
|
679
|
+
if own is None:
|
|
680
|
+
return None
|
|
681
|
+
for entry in manifest.get("steps") or []:
|
|
682
|
+
if not isinstance(entry, dict) or entry.get("reused"):
|
|
683
|
+
continue
|
|
684
|
+
files = entry.get("files") or []
|
|
685
|
+
if own in files:
|
|
686
|
+
shots = shots_for_file(entry.get("shots"), own, files)
|
|
687
|
+
if shots:
|
|
688
|
+
return shots
|
|
689
|
+
return None
|
|
690
|
+
|
|
691
|
+
|
|
692
|
+
def record_kept_shots(directory, file_name, shots):
|
|
693
|
+
"""Write or update the manifest sidecar beside a kept asset so
|
|
694
|
+
`shots_beside` can read the shot boundaries the source run recorded for
|
|
695
|
+
it (#393).
|
|
696
|
+
|
|
697
|
+
Keeping a file copies its bytes but not the run directory it lived in,
|
|
698
|
+
so a join's shot records - `pair_audio`'s picture is unchanged, but
|
|
699
|
+
nothing carried them past `keep_output` - were unreachable from the
|
|
700
|
+
asset and every probe saw `shots_source: "none"`. One manifest per
|
|
701
|
+
directory, keyed by file name, in the same shape a run's own
|
|
702
|
+
`manifest.json` uses, so the existing manifest-reading path (used by
|
|
703
|
+
both outputs and assets) picks it up with no change of its own. A
|
|
704
|
+
re-keep replaces the entry for that name rather than leaving a stale
|
|
705
|
+
one from a differently-shot source; `shots` of None or [] removes it.
|
|
706
|
+
"""
|
|
707
|
+
manifest_path = os.path.join(directory, MANIFEST_FILE_NAME)
|
|
708
|
+
manifest = _read_manifest(directory) or {}
|
|
709
|
+
steps = [
|
|
710
|
+
entry
|
|
711
|
+
for entry in manifest.get("steps") or []
|
|
712
|
+
if not (isinstance(entry, dict) and entry.get("files") == [file_name])
|
|
713
|
+
]
|
|
714
|
+
if shots:
|
|
715
|
+
steps.append({"step": "keep_output", "files": [file_name], "shots": shots})
|
|
716
|
+
if not steps:
|
|
717
|
+
try:
|
|
718
|
+
os.remove(manifest_path)
|
|
719
|
+
except OSError:
|
|
720
|
+
pass
|
|
721
|
+
return
|
|
722
|
+
manifest["steps"] = steps
|
|
723
|
+
with open(manifest_path, "w") as handle:
|
|
724
|
+
json.dump(manifest, handle)
|
|
725
|
+
|
|
726
|
+
|
|
727
|
+
# How far up from a file its run's manifest can sit: the run directory, a
|
|
728
|
+
# subfolder (`final/`) and the subfolder's own nesting, which
|
|
729
|
+
# SUBFOLDER_PATTERN caps well below this
|
|
730
|
+
MANIFEST_SEARCH_DEPTH = 8
|
|
731
|
+
|
|
732
|
+
|
|
733
|
+
def shots_beside(path):
|
|
734
|
+
"""The shots a run's manifest records for a file named by its absolute path.
|
|
735
|
+
|
|
736
|
+
`recorded_shots` needs the output root and a gallery name; a task that
|
|
737
|
+
was handed a resolved `output:` path has neither, only the file. The run
|
|
738
|
+
directory is the nearest parent holding a manifest, and the file's name
|
|
739
|
+
inside it is what the manifest records. None when no manifest is found
|
|
740
|
+
within MANIFEST_SEARCH_DEPTH parents, or it records no shots for the
|
|
741
|
+
file.
|
|
742
|
+
"""
|
|
743
|
+
from .shots import shots_for_file
|
|
744
|
+
|
|
745
|
+
path = os.path.abspath(path)
|
|
746
|
+
run_dir = os.path.dirname(path)
|
|
747
|
+
for _ in range(MANIFEST_SEARCH_DEPTH):
|
|
748
|
+
if os.path.isfile(os.path.join(run_dir, MANIFEST_FILE_NAME)):
|
|
749
|
+
break
|
|
750
|
+
parent = os.path.dirname(run_dir)
|
|
751
|
+
if parent == run_dir:
|
|
752
|
+
return None
|
|
753
|
+
run_dir = parent
|
|
754
|
+
else:
|
|
755
|
+
return None
|
|
756
|
+
manifest = _read_manifest(run_dir)
|
|
757
|
+
if manifest is None:
|
|
758
|
+
return None
|
|
759
|
+
own = os.path.relpath(path, run_dir).replace(os.sep, "/")
|
|
760
|
+
for entry in manifest.get("steps") or []:
|
|
761
|
+
if not isinstance(entry, dict) or entry.get("reused"):
|
|
762
|
+
continue
|
|
763
|
+
files = entry.get("files") or []
|
|
764
|
+
if own in files:
|
|
765
|
+
shots = shots_for_file(entry.get("shots"), own, files)
|
|
766
|
+
if shots:
|
|
767
|
+
return shots
|
|
768
|
+
return None
|