diffusers-workflow 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffusers_workflow-0.4.0.dist-info/METADATA +318 -0
- diffusers_workflow-0.4.0.dist-info/RECORD +260 -0
- diffusers_workflow-0.4.0.dist-info/WHEEL +5 -0
- diffusers_workflow-0.4.0.dist-info/entry_points.txt +7 -0
- diffusers_workflow-0.4.0.dist-info/licenses/LICENSE +201 -0
- diffusers_workflow-0.4.0.dist-info/top_level.txt +2 -0
- dw/__init__.py +440 -0
- dw/adapter_compatibility.py +226 -0
- dw/arguments.py +1231 -0
- dw/assessment_rules.py +159 -0
- dw/assets.py +130 -0
- dw/cache_blocks.json +16 -0
- dw/cache_blocks.py +146 -0
- dw/community_pipelines/pipeline_flux_rf_inversion.py +1184 -0
- dw/content_types.py +150 -0
- dw/dissolve_frame_errors.py +121 -0
- dw/docs/ACCELERATION.md +352 -0
- dw/docs/AGENT_LOOP.md +95 -0
- dw/docs/DEPENDENCIES.md +91 -0
- dw/docs/IP_ADAPTER.md +109 -0
- dw/docs/LORAS.md +131 -0
- dw/docs/MCP.md +517 -0
- dw/docs/PROMPT_WEIGHTING.md +78 -0
- dw/docs/QUANTIZATION.md +230 -0
- dw/docs/RECIPES_24GB.md +201 -0
- dw/docs/RELEASING.md +195 -0
- dw/docs/REMOTE.md +140 -0
- dw/docs/REPL_COMMANDS.md +121 -0
- dw/docs/REPL_WORKER_GUIDE.md +51 -0
- dw/docs/SECURITY.md +272 -0
- dw/docs/SECURITY_QUICKREF.md +112 -0
- dw/docs/SERVER.md +679 -0
- dw/docs/TASKS.md +1741 -0
- dw/docs/TESTING.md +71 -0
- dw/docs/WORKFLOW_GUIDE.md +2038 -0
- dw/docs/WORKSPACES.md +316 -0
- dw/download_watch.py +335 -0
- dw/elision.py +306 -0
- dw/events.py +275 -0
- dw/for_each.py +409 -0
- dw/host_memory.py +258 -0
- dw/host_memory_projection.py +230 -0
- dw/hub_cache.py +432 -0
- dw/introspection.py +1228 -0
- dw/kernel_availability.py +208 -0
- dw/locations.py +599 -0
- dw/log_setup.py +45 -0
- dw/loudness.py +82 -0
- dw/media_audio.py +217 -0
- dw/media_frames.py +367 -0
- dw/media_info.py +297 -0
- dw/pipeline_processors/chain.py +821 -0
- dw/pipeline_processors/config_objects.py +237 -0
- dw/pipeline_processors/pipeline.py +2297 -0
- dw/pipeline_processors/remote.py +46 -0
- dw/plan.py +920 -0
- dw/previous_results.py +411 -0
- dw/probe_paths.py +59 -0
- dw/prompt_schema.json +48 -0
- dw/prompt_weighting.py +378 -0
- dw/prompts.py +159 -0
- dw/realize.py +250 -0
- dw/reference_limits.py +215 -0
- dw/reference_names.py +125 -0
- dw/repl.py +338 -0
- dw/repl_commands.py +836 -0
- dw/repl_worker.py +159 -0
- dw/result.py +1720 -0
- dw/result_fps.py +82 -0
- dw/run.py +162 -0
- dw/runs.py +768 -0
- dw/scalar_result_validation.py +97 -0
- dw/schema.py +283 -0
- dw/security.py +1038 -0
- dw/select_validation.py +115 -0
- dw/serve.py +277 -0
- dw/server/__init__.py +2 -0
- dw/server/app.py +4586 -0
- dw/server/assess.py +132 -0
- dw/server/catalog_shape.py +487 -0
- dw/server/enhancers.py +129 -0
- dw/server/exports.py +480 -0
- dw/server/guides.py +257 -0
- dw/server/jobs.py +1561 -0
- dw/server/mcp_mount.py +95 -0
- dw/server/netinfo.py +124 -0
- dw/server/observed_cost.py +379 -0
- dw/server/sysinfo.py +71 -0
- dw/server/ui/assets/abap-08VXUWAP.js +1 -0
- dw/server/ui/assets/apex-BWPQTe0t.js +1 -0
- dw/server/ui/assets/azcli-Bc_sGQ0U.js +1 -0
- dw/server/ui/assets/bat-i0X4ZdIN.js +1 -0
- dw/server/ui/assets/bicep-B5-_aFwp.js +2 -0
- dw/server/ui/assets/cameligo-DMUM7wLl.js +1 -0
- dw/server/ui/assets/clojure-Cm7r79vr.js +1 -0
- dw/server/ui/assets/codicon-Brq4_Ui5.ttf +0 -0
- dw/server/ui/assets/coffee-Ba7i2nA0.js +1 -0
- dw/server/ui/assets/cpp-C7h46wYY.js +1 -0
- dw/server/ui/assets/csharp-BKxtCVv1.js +1 -0
- dw/server/ui/assets/csp-bTuwJoIa.js +1 -0
- dw/server/ui/assets/css-DIMkf-bt.js +3 -0
- dw/server/ui/assets/css.worker-B3ciXF_0.js +93 -0
- dw/server/ui/assets/cssMode-CPznxfY8.js +1 -0
- dw/server/ui/assets/cypher-CVaqCwHa.js +1 -0
- dw/server/ui/assets/dart-onAF5SnQ.js +1 -0
- dw/server/ui/assets/dockerfile-DZFCIeNp.js +1 -0
- dw/server/ui/assets/ecl-D05T4iGw.js +1 -0
- dw/server/ui/assets/editor-jjEx9u7D.css +1 -0
- dw/server/ui/assets/editor.api-CpWcotrd.js +847 -0
- dw/server/ui/assets/editor.worker-q-txB4vs.js +30 -0
- dw/server/ui/assets/elixir-6RTg0lbw.js +1 -0
- dw/server/ui/assets/flow9-C5_-GSwl.js +1 -0
- dw/server/ui/assets/freemarker2-CXtRM8N4.js +3 -0
- dw/server/ui/assets/fsharp-C8Ef5oNN.js +1 -0
- dw/server/ui/assets/go-C-y9NEjX.js +1 -0
- dw/server/ui/assets/graphql-fmXr3nnJ.js +1 -0
- dw/server/ui/assets/handlebars-N7x-6NMY.js +1 -0
- dw/server/ui/assets/hcl-CpzslTdj.js +1 -0
- dw/server/ui/assets/html-PhsdjHSr.js +1 -0
- dw/server/ui/assets/html.worker-C93Ht9o9.js +506 -0
- dw/server/ui/assets/htmlMode-Dgj0SEok.js +1 -0
- dw/server/ui/assets/index-3Vw6WAPW.css +1 -0
- dw/server/ui/assets/index-DgrYhQd9.js +43 -0
- dw/server/ui/assets/ini-sBoK_t0W.js +1 -0
- dw/server/ui/assets/java-BEtHBSE6.js +1 -0
- dw/server/ui/assets/javascript-BJqN9Qhv.js +1 -0
- dw/server/ui/assets/json.worker-B2V3pomh.js +62 -0
- dw/server/ui/assets/jsonMode-DbM4SWSv.js +7 -0
- dw/server/ui/assets/julia-Bri6UV-V.js +1 -0
- dw/server/ui/assets/kotlin-BOotOW0E.js +1 -0
- dw/server/ui/assets/less-B9JPFI3C.js +2 -0
- dw/server/ui/assets/lexon-CfSJPG6W.js +1 -0
- dw/server/ui/assets/liquid-BWr8lEc4.js +1 -0
- dw/server/ui/assets/lspLanguageFeatures-C1iGuDyZ.js +4 -0
- dw/server/ui/assets/lua-CsQS60Ue.js +1 -0
- dw/server/ui/assets/m3-D-oSqn_W.js +1 -0
- dw/server/ui/assets/markdown-Cimd5fb3.js +1 -0
- dw/server/ui/assets/mdx-DAdMi_0p.js +1 -0
- dw/server/ui/assets/mips-CIPQ_RoX.js +1 -0
- dw/server/ui/assets/monaco--ixms01u.css +1 -0
- dw/server/ui/assets/monaco-BGCeEqaw.js +56 -0
- dw/server/ui/assets/msdax-DauUninz.js +1 -0
- dw/server/ui/assets/mysql-SOo6toE5.js +1 -0
- dw/server/ui/assets/objective-c-FvmIjYaQ.js +1 -0
- dw/server/ui/assets/pascal-DrH0SRf2.js +1 -0
- dw/server/ui/assets/pascaligo-D-ptJ9y-.js +1 -0
- dw/server/ui/assets/perl-oz_6vUea.js +1 -0
- dw/server/ui/assets/pgsql-DTj74zXo.js +1 -0
- dw/server/ui/assets/php-nr791fC2.js +1 -0
- dw/server/ui/assets/pla-CopQ2nXW.js +1 -0
- dw/server/ui/assets/postiats-43DmfD33.js +1 -0
- dw/server/ui/assets/powerquery-D3hlyOfw.js +1 -0
- dw/server/ui/assets/powershell-DmHpPYUd.js +1 -0
- dw/server/ui/assets/protobuf-C531GsRP.js +2 -0
- dw/server/ui/assets/pug-Z5eAx3Zn.js +1 -0
- dw/server/ui/assets/python-Bcn70HdC.js +1 -0
- dw/server/ui/assets/qsharp-DkqhCAOL.js +1 -0
- dw/server/ui/assets/r-BwWrilGY.js +1 -0
- dw/server/ui/assets/razor-D1HmNnby.js +1 -0
- dw/server/ui/assets/redis-ClamHrr6.js +1 -0
- dw/server/ui/assets/redshift-DT7zqm-g.js +1 -0
- dw/server/ui/assets/restructuredtext-BYgofb2h.js +1 -0
- dw/server/ui/assets/ruby-DezsRK8O.js +1 -0
- dw/server/ui/assets/rust-DdL9SqIa.js +1 -0
- dw/server/ui/assets/sb-CcwsVR0C.js +1 -0
- dw/server/ui/assets/scala-DHpiXF5c.js +1 -0
- dw/server/ui/assets/scheme-BeGwcela.js +1 -0
- dw/server/ui/assets/scss-gp-XZpBa.js +3 -0
- dw/server/ui/assets/shell-CC2rA5mh.js +1 -0
- dw/server/ui/assets/solidity-BEEn4gHE.js +1 -0
- dw/server/ui/assets/sophia-CRfGWb83.js +1 -0
- dw/server/ui/assets/sparql-D_Lu-MrJ.js +1 -0
- dw/server/ui/assets/sql-NEE52Syq.js +1 -0
- dw/server/ui/assets/st-DbInun42.js +1 -0
- dw/server/ui/assets/swift-Bxkupp3x.js +1 -0
- dw/server/ui/assets/systemverilog-Bz4Y3fRF.js +1 -0
- dw/server/ui/assets/tcl-DISqw1ZD.js +1 -0
- dw/server/ui/assets/ts.worker-D7T1-Ig5.js +67738 -0
- dw/server/ui/assets/tsMode-D6u0XmOW.js +11 -0
- dw/server/ui/assets/twig-De2hgUGE.js +1 -0
- dw/server/ui/assets/typescript-BU6v-LMV.js +1 -0
- dw/server/ui/assets/typespec-B8J7ngcE.js +1 -0
- dw/server/ui/assets/vb-DV3o63ZY.js +1 -0
- dw/server/ui/assets/wgsl-DpFanUEy.js +298 -0
- dw/server/ui/assets/workers-Cn7cTUKr.js +1 -0
- dw/server/ui/assets/xml--0LP2Lwk.js +1 -0
- dw/server/ui/assets/yaml-mpBg9jnt.js +1 -0
- dw/server/ui/index.html +17 -0
- dw/server/updater.py +192 -0
- dw/settings.py +98 -0
- dw/shot_span_preflight.py +116 -0
- dw/shots.py +359 -0
- dw/slice_preflight.py +148 -0
- dw/step.py +187 -0
- dw/step_cache.py +442 -0
- dw/subfolders.py +107 -0
- dw/task_domains.py +307 -0
- dw/tasks/assess.py +826 -0
- dw/tasks/audio_transcription.py +88 -0
- dw/tasks/audio_utils.py +1862 -0
- dw/tasks/background_remover.py +43 -0
- dw/tasks/borders.py +113 -0
- dw/tasks/compose_text.py +74 -0
- dw/tasks/concat_videos.py +300 -0
- dw/tasks/depth_estimator.py +54 -0
- dw/tasks/diffusion_upscale.py +109 -0
- dw/tasks/dissolve_videos.py +342 -0
- dw/tasks/format_messages.py +24 -0
- dw/tasks/gather.py +173 -0
- dw/tasks/grade.py +97 -0
- dw/tasks/image_to_text.py +43 -0
- dw/tasks/image_utils.py +764 -0
- dw/tasks/interpolate_frames.py +252 -0
- dw/tasks/judge.py +68 -0
- dw/tasks/model_cache.py +55 -0
- dw/tasks/pair_audio.py +268 -0
- dw/tasks/qr_code.py +19 -0
- dw/tasks/restore_faces.py +175 -0
- dw/tasks/rife_model.py +192 -0
- dw/tasks/segment.py +121 -0
- dw/tasks/select.py +111 -0
- dw/tasks/speech_generation.py +228 -0
- dw/tasks/stabilize.py +129 -0
- dw/tasks/task.py +920 -0
- dw/tasks/tensor_image.py +57 -0
- dw/tasks/text_generation.py +169 -0
- dw/tasks/text_sections.py +80 -0
- dw/tasks/upscale.py +203 -0
- dw/tasks/video_utils.py +624 -0
- dw/tasks/zoe_depth.py +71 -0
- dw/teacache.py +381 -0
- dw/teacache_models.json +99 -0
- dw/test.py +29 -0
- dw/type_helpers.py +231 -0
- dw/validate.py +68 -0
- dw/variable_constraints.py +444 -0
- dw/variables.py +443 -0
- dw/video_extensions.py +141 -0
- dw/vram_estimate.py +116 -0
- dw/worker.py +764 -0
- dw/workflow.py +2007 -0
- dw/workflow_schema.json +1346 -0
- dw/workflow_sources.py +383 -0
- dw/workflows/h3_context_ir.json +57 -0
- dw/workflows/test.json +31 -0
- dw/workspace.py +730 -0
- dw_mcp/__init__.py +6 -0
- dw_mcp/__main__.py +133 -0
- dw_mcp/assets.py +336 -0
- dw_mcp/authoring.py +114 -0
- dw_mcp/catalog.py +360 -0
- dw_mcp/client.py +486 -0
- dw_mcp/diagnose.py +371 -0
- dw_mcp/exports.py +84 -0
- dw_mcp/guides.py +35 -0
- dw_mcp/media.py +638 -0
- dw_mcp/models.py +97 -0
- dw_mcp/prompts.py +104 -0
- dw_mcp/server.py +1343 -0
- dw_mcp/workspaces.py +212 -0
dw/docs/WORKSPACES.md
ADDED
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
# Workspaces
|
|
2
|
+
|
|
3
|
+
A workspace is the directory your own content lives in: the workflows you
|
|
4
|
+
write, the prompt library they reference, the assets they read, and the files
|
|
5
|
+
they generate.
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
<workspace>/
|
|
9
|
+
workflows/ your workflows
|
|
10
|
+
prompts/ the stored prompt library ('prompt:' references)
|
|
11
|
+
assets/ input media
|
|
12
|
+
outputs/ generated files
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
This exists so that day-to-day work does not have to live inside a checkout of
|
|
16
|
+
this repository. The examples under the repo's `workflows/` are still examples
|
|
17
|
+
— a corpus to read, copy and run — but they are not where your own workflows
|
|
18
|
+
belong, and generated media does not belong in a source tree at all.
|
|
19
|
+
|
|
20
|
+
## Which directory is used
|
|
21
|
+
|
|
22
|
+
First match wins:
|
|
23
|
+
|
|
24
|
+
1. `--workspace <dir>` on `dw.run`, `dw.serve` (`config set workspace=` in the REPL)
|
|
25
|
+
2. the `DW_WORKSPACE` environment variable
|
|
26
|
+
3. `"workspace"` in `~/.diffusers_helper/settings.json`
|
|
27
|
+
4. the working directory, when it holds any of `workflows/`, `prompts/` or `outputs/`
|
|
28
|
+
5. `~/diffusers-workspace`
|
|
29
|
+
|
|
30
|
+
Rule 4 is why nothing changes when you work from a checkout: the repository
|
|
31
|
+
root holds all three, so it resolves to itself and every default lands exactly
|
|
32
|
+
where it always has. Only a working directory with none of those folders falls
|
|
33
|
+
through to the home workspace.
|
|
34
|
+
|
|
35
|
+
Nothing is created just by resolving. A command that is about to write creates
|
|
36
|
+
what it needs — `dw.run` creates its output directory, `dw.serve` creates the
|
|
37
|
+
workspace's `workflows/` so the UI has somewhere to save.
|
|
38
|
+
|
|
39
|
+
## Overriding one folder
|
|
40
|
+
|
|
41
|
+
The existing per-directory flags still work and each overrides exactly one
|
|
42
|
+
folder of the workspace:
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
python -m dw.run workflows/templates/text-to-image.json -o /mnt/big-disk/renders
|
|
46
|
+
python -m dw.serve --workspace ~/studio --output-dir /mnt/big-disk/renders
|
|
47
|
+
python -m dw.run some.json --prompt-dir ~/shared-prompts
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
`--output-dir` is the one people reach for most: video work fills disks, and
|
|
51
|
+
the outputs folder is the one worth putting on another volume.
|
|
52
|
+
|
|
53
|
+
## Working in a workspace
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
mkdir -p ~/studio/{workflows,prompts,assets,outputs}
|
|
57
|
+
export DW_WORKSPACE=~/studio
|
|
58
|
+
|
|
59
|
+
# or, standing, in ~/.diffusers_helper/settings.json
|
|
60
|
+
# { "workspace": "/home/you/studio" }
|
|
61
|
+
|
|
62
|
+
python -m dw.serve # serves ~/studio
|
|
63
|
+
python -m dw.run ~/studio/workflows/x.json
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
An example from a checkout still runs by path, and writes into the workspace's
|
|
67
|
+
outputs:
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
DW_WORKSPACE=~/studio python -m dw.run ~/src/diffusers-workflow/workflows/templates/text-to-image.json
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
The server reports what it resolved at `GET /api/server`, under
|
|
74
|
+
`directories.workspace` alongside the three folder paths.
|
|
75
|
+
|
|
76
|
+
## The prompt library
|
|
77
|
+
|
|
78
|
+
`prompt:` references resolve to the workspace's `prompts/` when a workspace was
|
|
79
|
+
named explicitly (rules 1–3 above, or `DW_PROMPT_DIR`, which still wins over
|
|
80
|
+
everything). A workspace that was merely inferred from the working directory
|
|
81
|
+
does not preempt the older discovery — `./prompts`, then the nearest `prompts/`
|
|
82
|
+
above the workflow file — so a repository workflow keeps reaching the library
|
|
83
|
+
it lives beside. When the workspace is explicit, its `prompts/` becomes the library
|
|
84
|
+
even if it does not exist yet, so a checkout's `./prompts` is no longer found once
|
|
85
|
+
a standing workspace setting (like `DW_WORKSPACE` or `"workspace"` in settings.json)
|
|
86
|
+
is in place; `--prompt-dir` and `DW_PROMPT_DIR` still override it. This follows
|
|
87
|
+
the "explicit wins" rule. See [Prompt References](WORKFLOW_GUIDE.md#prompt-references).
|
|
88
|
+
|
|
89
|
+
## Assets
|
|
90
|
+
|
|
91
|
+
`assets/` is the input-media library. A workflow argument written as
|
|
92
|
+
`asset:name.ext` (or `asset:folder/name.ext`) resolves to that file's path,
|
|
93
|
+
rooted at the library rather than at the workflow file — so a workflow and the
|
|
94
|
+
media it reads no longer have to sit in the same folder. `--asset-dir` and
|
|
95
|
+
`DW_ASSET_DIR` override the folder, and browser uploads land in
|
|
96
|
+
`assets/uploads/`, coming back as `asset:uploads/<name>` (and served for
|
|
97
|
+
preview under `/inputs/`, since the SPA's own bundles own `/assets/`).
|
|
98
|
+
|
|
99
|
+
A generated file becomes an input the same way: **Keep as asset** in the
|
|
100
|
+
gallery (`POST /api/assets/keep`, `keep_output` over MCP) links or copies it
|
|
101
|
+
out of `outputs/` into `assets/` under a name you choose, so a later workflow
|
|
102
|
+
carries `asset:<name>` rather than a run id that pruning would break. The copy
|
|
103
|
+
stays inside the workspace — a hard link where the filesystem allows one, so
|
|
104
|
+
keeping one frame of a large render costs no second copy of it. See
|
|
105
|
+
[Asset References](WORKFLOW_GUIDE.md#asset-references).
|
|
106
|
+
|
|
107
|
+
## Where workflows are read from, and written to
|
|
108
|
+
|
|
109
|
+
The server reads workflows from a search path and writes them to exactly one
|
|
110
|
+
place — the front:
|
|
111
|
+
|
|
112
|
+
```
|
|
113
|
+
<workspace>/workflows/ yours, writable — every save lands here
|
|
114
|
+
<--examples-dir> read-only, repeatable
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
A name found in an earlier root shadows the same name in a later one, so a
|
|
118
|
+
workspace copy of an example is the one that runs. Reads — listing, opening,
|
|
119
|
+
downloading, validating, running — span the whole path. Saves and deletes do
|
|
120
|
+
not: `PUT` always writes into the writable root, and deleting something from a
|
|
121
|
+
read-only root is refused with a 403 that says where it came from.
|
|
122
|
+
|
|
123
|
+
That makes "open an example, change it, save" do the obvious thing: the copy
|
|
124
|
+
lands in your library and shadows the example from then on, and the example
|
|
125
|
+
itself is never touched. It is also what stops an agent's saves landing in a
|
|
126
|
+
checkout — point `--workflow-dir` (or `--workspace`) at your own directory and
|
|
127
|
+
the repository's workflows at `--examples-dir`:
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
python -m dw.serve --workspace ~/studio --examples-dir ~/src/diffusers-workflow/workflows
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
`GET /api/workflows` reports the path as `sources` and tags every entry with
|
|
134
|
+
its `origin` and `writable`, which is how the UI knows to hide delete and how
|
|
135
|
+
an MCP client can tell what it may change.
|
|
136
|
+
|
|
137
|
+
### The prompts and assets an examples tree brings with it
|
|
138
|
+
|
|
139
|
+
An example workflow references the prompts and media that live beside its
|
|
140
|
+
tree, not the ones in your workspace, so each `--examples-dir` puts those on
|
|
141
|
+
the back of the two libraries as well: the `prompts/` and `assets/` folders
|
|
142
|
+
beside the directory (or inside it, if that is where they are). Both libraries
|
|
143
|
+
work exactly like the workflow path — your workspace's own is searched first
|
|
144
|
+
and a name there shadows the example's, reads span everything, and writes only
|
|
145
|
+
ever reach your own:
|
|
146
|
+
|
|
147
|
+
```
|
|
148
|
+
<workspace>/prompts/ yours, writable — every save lands here
|
|
149
|
+
<--examples-dir>/../prompts read-only
|
|
150
|
+
|
|
151
|
+
<workspace>/assets/ yours, writable — uploads and "keep as asset" land here
|
|
152
|
+
<--examples-dir>/../assets read-only
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
So the command above makes `workflows/models/flux-dev.json`'s
|
|
156
|
+
`"prompt:flux/biomechanical_daffodil"` resolve out of the checkout, without
|
|
157
|
+
copying the prompt library into the workspace. `GET /api/prompts` and
|
|
158
|
+
`GET /api/assets` report the roots as `prompt_dirs` / `asset_dirs` and tag
|
|
159
|
+
each entry with its `origin`; deleting a prompt that came from a read-only
|
|
160
|
+
library is refused with a 403, and saving one writes a copy into your
|
|
161
|
+
workspace the way saving an example workflow does.
|
|
162
|
+
|
|
163
|
+
The packaged workflows in `dw/workflows/` are deliberately *not* on the path.
|
|
164
|
+
They are the pieces a `builtin:` sub-workflow step names, resolved by the
|
|
165
|
+
engine where that step is read — not workflows to browse or run on their own.
|
|
166
|
+
|
|
167
|
+
## Runs
|
|
168
|
+
|
|
169
|
+
Each execution writes its own directory under the output folder, named by the
|
|
170
|
+
workflow and the run:
|
|
171
|
+
|
|
172
|
+
```
|
|
173
|
+
outputs/
|
|
174
|
+
ltx2/Gyre/
|
|
175
|
+
20260905-181530-a1b2c3d4/
|
|
176
|
+
Gyre-still.0-0.0.png
|
|
177
|
+
Gyre-video.1-0.0.mp4
|
|
178
|
+
manifest.json
|
|
179
|
+
workflow.json
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
The folder is the workflow's identity — its path under a `workflows/` tree
|
|
183
|
+
when it has one, its file name otherwise, its `id` for an inline definition.
|
|
184
|
+
The run id is a timestamp plus a short digest of what actually ran, so two
|
|
185
|
+
runs of the same workflow sort by time and a rerun of an edited workflow is
|
|
186
|
+
visibly different; a second run of the same spec in the same second takes a
|
|
187
|
+
counter rather than sharing a directory.
|
|
188
|
+
|
|
189
|
+
`workflow.json` is the realized workflow — the definition with this run's
|
|
190
|
+
arguments, seed and stored prompts pinned into it, so the directory reproduces
|
|
191
|
+
itself. `manifest.json` points at it and lists which prompts were inlined.
|
|
192
|
+
|
|
193
|
+
`manifest.json` records the run beside what it made — status, seed, arguments,
|
|
194
|
+
device, dw version, and each step's files, named relative to the directory so
|
|
195
|
+
it keeps describing itself if you move or copy it. It is written even when a
|
|
196
|
+
run fails part way, since the files it did write are on disk either way. A
|
|
197
|
+
sub-workflow is part of its parent's run: it writes into the same directory and
|
|
198
|
+
rolls up into the same manifest.
|
|
199
|
+
|
|
200
|
+
An unchanged rerun still reuses the step cache: it writes no new files and its
|
|
201
|
+
manifest reports the earlier run's, marked `"reused": true`. The cache is
|
|
202
|
+
validated against the output *root* a run writes into, which is the pinned
|
|
203
|
+
workspace's own `outputs/` - so it is per workspace, not per workflow alone.
|
|
204
|
+
A run in one workspace does not make `validate_workflow`'s
|
|
205
|
+
`plan.cached_steps` come back nonzero for a matching run sitting in another
|
|
206
|
+
workspace, and deleting a workspace drops its cache entries along with its
|
|
207
|
+
`outputs/` directory. The cache itself is also per *process*: entries are
|
|
208
|
+
held in memory by the running server, not read back from `outputs/`, so a
|
|
209
|
+
`dw.serve` restart empties it even though every run directory is still on
|
|
210
|
+
disk - a `plan.cached_steps` of 0 right after a restart is expected, not a
|
|
211
|
+
lost run.
|
|
212
|
+
|
|
213
|
+
A later workflow names what an earlier run made with an `output:` reference —
|
|
214
|
+
`output:ltx2/Gyre/latest/Gyre-still.0-0.0.png` — so a multi-stage pipeline no
|
|
215
|
+
longer needs files copied back by hand. See
|
|
216
|
+
[Output References](WORKFLOW_GUIDE.md#output-references).
|
|
217
|
+
|
|
218
|
+
To keep the previous layout — everything at the output root, with only a
|
|
219
|
+
`workflows/`-mirroring subfolder — use `--output-layout flat`, `DW_OUTPUT_LAYOUT=flat`,
|
|
220
|
+
or `"output_layout": "flat"` in settings. Scripts that glob the output directory
|
|
221
|
+
are the reason to.
|
|
222
|
+
|
|
223
|
+
## Several workspaces on one server
|
|
224
|
+
|
|
225
|
+
Everything above describes one workspace, which is all `dw.run` and the REPL
|
|
226
|
+
ever see. `dw.serve` goes one step further: the workspace root can hold
|
|
227
|
+
several, and a client picks which one it is working in.
|
|
228
|
+
|
|
229
|
+
```
|
|
230
|
+
<workspace root>/
|
|
231
|
+
workflows/ assets/ outputs/ <- the 'default' workspace
|
|
232
|
+
prompts/ <- shared by all of them
|
|
233
|
+
common/assets/ <- shared by all of them
|
|
234
|
+
studio/
|
|
235
|
+
workflows/ assets/ outputs/ <- the 'studio' workspace
|
|
236
|
+
scratch/
|
|
237
|
+
workflows/ assets/ outputs/ <- the 'scratch' workspace
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
The root's own three folders are the workspace named `default`, so a server
|
|
241
|
+
that has never heard of named workspaces behaves exactly as it did. A named
|
|
242
|
+
workspace is a sibling directory holding the same three folders — and *not* a
|
|
243
|
+
`prompts/`, because there is one prompt library: `prompt:` is shared by
|
|
244
|
+
reference, and a prompt duplicated per workspace would resolve to different
|
|
245
|
+
text depending on where a workflow happened to be saved. `workflows`,
|
|
246
|
+
`prompts`, `assets` and `outputs` are reserved names for that reason.
|
|
247
|
+
|
|
248
|
+
Two more names are reserved beside `workflows`, `prompts`, `assets` and
|
|
249
|
+
`outputs`: `exports` and `common`. `POST /api/jobs/{id}/export` gathers one
|
|
250
|
+
finished job into `<root>/exports/<job id>/`, and that folder is never mistaken
|
|
251
|
+
for a workspace.
|
|
252
|
+
|
|
253
|
+
**The shared asset library.** `common/assets/` is the one place an asset can
|
|
254
|
+
live that belongs to no single workspace. Assets are otherwise per workspace,
|
|
255
|
+
which is right for the inputs of one piece of work and wrong for a recurring
|
|
256
|
+
cast: a character's portrait and voice clip uploaded while making episode one
|
|
257
|
+
were invisible from the workspace episode four was made in, and the only way
|
|
258
|
+
through was to copy the files in. It sits on every workspace's asset search
|
|
259
|
+
path behind that workspace's own library, so:
|
|
260
|
+
|
|
261
|
+
- `asset:cast/priya.png` resolves in the workspace first, then in the shared
|
|
262
|
+
library, then in any read-only examples library — a workspace's own name
|
|
263
|
+
still shadows a shared one
|
|
264
|
+
- `GET /api/assets` spans all of them and tags each entry's `origin`:
|
|
265
|
+
`workspace`, `common`, or `examples`
|
|
266
|
+
- writes still land in the workspace unless they say otherwise:
|
|
267
|
+
`POST /api/uploads?shared=true`, `POST /api/assets/keep` with
|
|
268
|
+
`"shared": true`, and over MCP `upload_asset(..., shared=True)` /
|
|
269
|
+
`keep_output(..., shared=True)`
|
|
270
|
+
|
|
271
|
+
It holds assets only. A prompt is already shared, and workflows and outputs
|
|
272
|
+
belong to the work that made them.
|
|
273
|
+
|
|
274
|
+
This is what lets two agents share one GPU without sharing a namespace: each
|
|
275
|
+
takes a workspace, and neither can save over the other's workflows or delete
|
|
276
|
+
the other's renders.
|
|
277
|
+
|
|
278
|
+
**How a client picks one.** Every scoped route takes an optional
|
|
279
|
+
`?workspace=<name>`; omitting it means `default`, which is why every
|
|
280
|
+
pre-workspace call still means what it meant.
|
|
281
|
+
|
|
282
|
+
| Client | How |
|
|
283
|
+
| --- | --- |
|
|
284
|
+
| Web UI | The sidebar lists every workspace; the selected one is named in the hash (`#/ws/<name>/...`), so a link and a reload both land where they say. The choice is remembered in `localStorage` as a fallback for a route that names none (Shared, Server), and Server → Status adds a filter over the all-workspaces queue — job history spans every workspace there and says which one each job ran in |
|
|
285
|
+
| MCP | `list_workspaces`, then `use_workspace(name)`. It is a session default rather than an argument on each call, so switching is one visible step in the transcript instead of a flag that can be forgotten on the call where it mattered |
|
|
286
|
+
| HTTP | `?workspace=` on the route, or `"workspace"` in a `POST /api/jobs` body |
|
|
287
|
+
| Web UI (create/delete) | The sidebar's `+ new` creates one; a workspace's own Overview page deletes it (disabled for `default`) |
|
|
288
|
+
|
|
289
|
+
A job carries its own workflow, asset and output directories, so it stays in
|
|
290
|
+
the workspace it was submitted from however many others the server serves
|
|
291
|
+
while it runs — including through a rerun, and when its files are served back
|
|
292
|
+
from history.
|
|
293
|
+
|
|
294
|
+
**Creating and deleting.** `POST /api/workspaces` (`create_workspace` over
|
|
295
|
+
MCP) makes one; creating does not switch to it. Deleting removes everything in
|
|
296
|
+
it, so it refuses until acknowledged and answers first with what it would
|
|
297
|
+
remove — file counts and bytes per folder. The default workspace cannot be
|
|
298
|
+
deleted (it holds the shared prompt library, and there has to be somewhere to
|
|
299
|
+
work), nor can one with jobs still queued.
|
|
300
|
+
|
|
301
|
+
A workspace is a **namespace, not a security boundary**. The API token is
|
|
302
|
+
all-or-nothing: anything that can reach the server can name any workspace on
|
|
303
|
+
it. Use them to keep work apart, not to keep it private.
|
|
304
|
+
|
|
305
|
+
## Where this is going
|
|
306
|
+
|
|
307
|
+
The resolver, the workflow search path with writes confined to the writable
|
|
308
|
+
root, run directories with an on-disk manifest, `asset:` and `output:`
|
|
309
|
+
references, and server-side named workspaces are all implemented. A further
|
|
310
|
+
stage was designed but deliberately not built: a client-side workspace (a
|
|
311
|
+
laptop directory, under version control) that mirrors into a read-only
|
|
312
|
+
server workspace, so an agent could author offline and only push at submit
|
|
313
|
+
time. It stayed on the drawing board because source control of creative work
|
|
314
|
+
is not this project's job — that is already handled on the client, by the
|
|
315
|
+
user, with the tools they already use — which is what makes a mirroring
|
|
316
|
+
layer unnecessary rather than merely speculative.
|
dw/download_watch.py
ADDED
|
@@ -0,0 +1,335 @@
|
|
|
1
|
+
"""Makes a `from_pretrained` download visible on the run's event stream (#343).
|
|
2
|
+
|
|
3
|
+
`from_pretrained` downloads with no hook of its own, and a pull that runs
|
|
4
|
+
long looks identical to a hang: the phase-stall watchdog (dw/events.py) has
|
|
5
|
+
nothing but silence to report. This does not intercept or pre-fetch
|
|
6
|
+
anything - `from_pretrained` fetches exactly what it would have - it only
|
|
7
|
+
listens to the byte counts the download already reports while a `loading`
|
|
8
|
+
phase is in progress.
|
|
9
|
+
|
|
10
|
+
Two signals, and the larger is reported:
|
|
11
|
+
|
|
12
|
+
- The hub's own xet progress report (`XetDownloadProgressReporter`), which
|
|
13
|
+
carries the bytes *received from the network* as well as the bytes written
|
|
14
|
+
to disk. The network count is the one that matters: hf_xet buffers and
|
|
15
|
+
writes a file in order, and on lem held a 2.8 GB file at 67 MB on disk
|
|
16
|
+
while 700 MB had arrived, then wrote the rest at the very end. No location
|
|
17
|
+
on disk tells that apart from a hang; the transfer count does. Hooked
|
|
18
|
+
whether or not the hub's progress bars are displayed.
|
|
19
|
+
- The repo's cache directory, for a plain HTTP download (no hf_xet), which
|
|
20
|
+
writes its `.incomplete` file as bytes arrive. A published blob is a
|
|
21
|
+
symlink into the hub's shared store (`<cache>/blobs/xx/<sha>`), so the
|
|
22
|
+
size is taken through the link - measured beside it, the total fell by the
|
|
23
|
+
size of every file that finished (#343, a negative rate).
|
|
24
|
+
|
|
25
|
+
Both only ever count up, so `downloaded_bytes` never goes backwards.
|
|
26
|
+
|
|
27
|
+
A `download_progress` event fires about every EMIT_INTERVAL_SECONDS while a
|
|
28
|
+
download is underway. One reporting growth is progress. One reporting none
|
|
29
|
+
is not: it still fires, so `phase_detail` says `no bytes for 45s` instead of
|
|
30
|
+
repeating the last healthy rate, but the stall watchdog counts it as silence
|
|
31
|
+
and says `phase_stall` once bytes have been flat for its threshold - a
|
|
32
|
+
download and a hang no longer read the same.
|
|
33
|
+
|
|
34
|
+
A cancel during a download aborts it: the xet session is aborted (the hub's
|
|
35
|
+
own KeyboardInterrupt path, `abort_xet_session`) and a plain HTTP download is
|
|
36
|
+
stopped at its next chunk, and the load surfaces as `WorkflowCancelled`
|
|
37
|
+
rather than running to the end of a multi-gigabyte file first.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
import logging
|
|
41
|
+
import os
|
|
42
|
+
import threading
|
|
43
|
+
import time
|
|
44
|
+
|
|
45
|
+
from huggingface_hub.constants import HF_HUB_CACHE
|
|
46
|
+
from huggingface_hub.file_download import repo_folder_name
|
|
47
|
+
from huggingface_hub.utils import HFValidationError, validate_repo_id
|
|
48
|
+
|
|
49
|
+
from .events import WorkflowCancelled, get_context
|
|
50
|
+
|
|
51
|
+
logger = logging.getLogger("dw")
|
|
52
|
+
|
|
53
|
+
# How often the watcher re-measures, and the gap between emitted events -
|
|
54
|
+
# well under the phase-stall threshold (30s) so a real, ongoing download
|
|
55
|
+
# never trips it, but not so tight that a run emits an event per chunk.
|
|
56
|
+
CHECK_INTERVAL_SECONDS = 1.0
|
|
57
|
+
EMIT_INTERVAL_SECONDS = 5.0
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def is_watchable_repo_id(name):
|
|
61
|
+
"""Whether name is shaped like a Hugging Face repo id - a local path or
|
|
62
|
+
checkout is not something a cache directory watch means anything for."""
|
|
63
|
+
try:
|
|
64
|
+
validate_repo_id(name)
|
|
65
|
+
return True
|
|
66
|
+
except HFValidationError:
|
|
67
|
+
return False
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _blob_dir_size(blob_dir):
|
|
71
|
+
"""Bytes under a repo's blobs/ directory, counting a published blob
|
|
72
|
+
through its symlink into the shared store."""
|
|
73
|
+
total = 0
|
|
74
|
+
try:
|
|
75
|
+
with os.scandir(blob_dir) as entries:
|
|
76
|
+
for entry in entries:
|
|
77
|
+
try:
|
|
78
|
+
if entry.is_file():
|
|
79
|
+
total += entry.stat().st_size
|
|
80
|
+
except OSError:
|
|
81
|
+
# A blob renamed or published mid-scan is not a fault
|
|
82
|
+
continue
|
|
83
|
+
except FileNotFoundError:
|
|
84
|
+
return 0
|
|
85
|
+
return total
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
# The watches currently open and the hub hooks they share. One job runs at a
|
|
89
|
+
# time in the worker, but a hook is process-wide, so it is installed on the
|
|
90
|
+
# first open watch and removed with the last.
|
|
91
|
+
_lock = threading.Lock()
|
|
92
|
+
_active = []
|
|
93
|
+
_originals = {}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _note_hub_bytes(transferred, written):
|
|
97
|
+
with _lock:
|
|
98
|
+
watches = list(_active)
|
|
99
|
+
for watch_ in watches:
|
|
100
|
+
watch_._add_hub_bytes(transferred, written)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _any_cancelled():
|
|
104
|
+
with _lock:
|
|
105
|
+
return any(w._context.cancelled for w in _active)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _install_hooks():
|
|
109
|
+
try:
|
|
110
|
+
from huggingface_hub.utils import _xet_progress_reporting as xet_reporting
|
|
111
|
+
|
|
112
|
+
reporter = xet_reporting.XetDownloadProgressReporter
|
|
113
|
+
original = reporter.update_progress
|
|
114
|
+
last_seen = {}
|
|
115
|
+
|
|
116
|
+
def update_progress(self, group_report, *args, **kwargs):
|
|
117
|
+
# Runs on hf_xet's callback thread, which prints and swallows
|
|
118
|
+
# anything raised here - so nothing may be raised, and the
|
|
119
|
+
# hub's own bar update is guarded along with the counting
|
|
120
|
+
try:
|
|
121
|
+
key = id(self)
|
|
122
|
+
previous = last_seen.get(key, (0, 0))
|
|
123
|
+
transferred = group_report.total_transfer_bytes_completed
|
|
124
|
+
written = group_report.total_bytes_completed
|
|
125
|
+
last_seen[key] = (
|
|
126
|
+
max(previous[0], transferred),
|
|
127
|
+
max(previous[1], written),
|
|
128
|
+
)
|
|
129
|
+
_note_hub_bytes(
|
|
130
|
+
max(0, transferred - previous[0]), max(0, written - previous[1])
|
|
131
|
+
)
|
|
132
|
+
except Exception as e:
|
|
133
|
+
logger.debug(f"Download watch could not read xet progress: {e}")
|
|
134
|
+
try:
|
|
135
|
+
return original(self, group_report, *args, **kwargs)
|
|
136
|
+
except Exception as e:
|
|
137
|
+
logger.debug(f"xet progress update raised: {e}")
|
|
138
|
+
|
|
139
|
+
_patch(reporter, "update_progress", update_progress)
|
|
140
|
+
except (ImportError, AttributeError) as e:
|
|
141
|
+
logger.debug(f"No xet progress reporter to hook: {e}")
|
|
142
|
+
|
|
143
|
+
try:
|
|
144
|
+
from importlib import import_module
|
|
145
|
+
|
|
146
|
+
hub_tqdm = import_module("huggingface_hub.utils.tqdm").tqdm
|
|
147
|
+
original_update = hub_tqdm.update
|
|
148
|
+
|
|
149
|
+
def update(self, n=1):
|
|
150
|
+
# A plain HTTP download calls this once per chunk on the thread
|
|
151
|
+
# running from_pretrained - the one place that can stop it
|
|
152
|
+
if _any_cancelled():
|
|
153
|
+
raise WorkflowCancelled("Workflow run was cancelled")
|
|
154
|
+
return original_update(self, n)
|
|
155
|
+
|
|
156
|
+
_patch(hub_tqdm, "update", update)
|
|
157
|
+
except (ImportError, AttributeError) as e:
|
|
158
|
+
logger.debug(f"No hub progress bar to hook: {e}")
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _patch(owner, name, replacement):
|
|
162
|
+
# Remembers whether the class defined the method itself or inherited it,
|
|
163
|
+
# so removing the hook restores exactly what was there
|
|
164
|
+
_originals[(owner, name)] = vars(owner).get(name)
|
|
165
|
+
setattr(owner, name, replacement)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _remove_hooks():
|
|
169
|
+
for (owner, name), original in _originals.items():
|
|
170
|
+
if original is None:
|
|
171
|
+
delattr(owner, name)
|
|
172
|
+
else:
|
|
173
|
+
setattr(owner, name, original)
|
|
174
|
+
_originals.clear()
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _abort_xet_downloads():
|
|
178
|
+
try:
|
|
179
|
+
from huggingface_hub.utils._xet import abort_xet_session
|
|
180
|
+
|
|
181
|
+
abort_xet_session()
|
|
182
|
+
except Exception as e:
|
|
183
|
+
logger.debug(f"Could not abort the xet session: {e}")
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
class DownloadWatch:
|
|
187
|
+
"""Context manager that reports one repo's download on the active run
|
|
188
|
+
context for as long as it is open.
|
|
189
|
+
|
|
190
|
+
Not a download itself - `from_pretrained` runs unmodified inside the
|
|
191
|
+
`with` block. A load that downloads nothing (every file cached) emits
|
|
192
|
+
nothing, same as no watch at all.
|
|
193
|
+
"""
|
|
194
|
+
|
|
195
|
+
def __init__(self, repo_id, context, cache_dir=None, repo_type="model"):
|
|
196
|
+
self._repo_id = repo_id
|
|
197
|
+
self._context = context
|
|
198
|
+
resolved_cache_dir = cache_dir or HF_HUB_CACHE
|
|
199
|
+
self._blob_dir = os.path.join(
|
|
200
|
+
resolved_cache_dir,
|
|
201
|
+
repo_folder_name(repo_id=repo_id, repo_type=repo_type),
|
|
202
|
+
"blobs",
|
|
203
|
+
)
|
|
204
|
+
self._stop = threading.Event()
|
|
205
|
+
self._thread = None
|
|
206
|
+
self._counts_lock = threading.Lock()
|
|
207
|
+
self._transferred = 0
|
|
208
|
+
self._written = 0
|
|
209
|
+
self._aborted = False
|
|
210
|
+
|
|
211
|
+
def _add_hub_bytes(self, transferred, written):
|
|
212
|
+
with self._counts_lock:
|
|
213
|
+
self._transferred += transferred
|
|
214
|
+
self._written += written
|
|
215
|
+
|
|
216
|
+
def _hub_bytes(self):
|
|
217
|
+
with self._counts_lock:
|
|
218
|
+
return max(self._transferred, self._written)
|
|
219
|
+
|
|
220
|
+
def __enter__(self):
|
|
221
|
+
with _lock:
|
|
222
|
+
if not _active:
|
|
223
|
+
_install_hooks()
|
|
224
|
+
_active.append(self)
|
|
225
|
+
self._thread = threading.Thread(
|
|
226
|
+
target=self._run, daemon=True, name="dw-download-watch"
|
|
227
|
+
)
|
|
228
|
+
self._thread.start()
|
|
229
|
+
return self
|
|
230
|
+
|
|
231
|
+
def __exit__(self, exc_type, exc, traceback):
|
|
232
|
+
self._stop.set()
|
|
233
|
+
if self._thread is not None:
|
|
234
|
+
self._thread.join(timeout=CHECK_INTERVAL_SECONDS * 2)
|
|
235
|
+
with _lock:
|
|
236
|
+
if self in _active:
|
|
237
|
+
_active.remove(self)
|
|
238
|
+
if not _active:
|
|
239
|
+
_remove_hooks()
|
|
240
|
+
if (
|
|
241
|
+
exc is not None
|
|
242
|
+
and not isinstance(exc, WorkflowCancelled)
|
|
243
|
+
and self._context.cancelled
|
|
244
|
+
):
|
|
245
|
+
# The abort surfaces as whatever the download library raises
|
|
246
|
+
# (hf_xet: "RuntimeError: Operation cancelled") - it is the
|
|
247
|
+
# cancel, not a load failure, and must read as one
|
|
248
|
+
raise WorkflowCancelled("Workflow run was cancelled") from exc
|
|
249
|
+
return False
|
|
250
|
+
|
|
251
|
+
def _run(self):
|
|
252
|
+
baseline = _blob_dir_size(self._blob_dir)
|
|
253
|
+
disk_high_water = 0
|
|
254
|
+
downloaded = 0
|
|
255
|
+
emitted = 0
|
|
256
|
+
last_emit_at = time.monotonic()
|
|
257
|
+
last_change_at = last_emit_at
|
|
258
|
+
while not self._stop.wait(CHECK_INTERVAL_SECONDS):
|
|
259
|
+
try:
|
|
260
|
+
if self._context.cancelled and not self._aborted:
|
|
261
|
+
self._aborted = True
|
|
262
|
+
_abort_xet_downloads()
|
|
263
|
+
disk_high_water = max(
|
|
264
|
+
disk_high_water, _blob_dir_size(self._blob_dir) - baseline
|
|
265
|
+
)
|
|
266
|
+
now = time.monotonic()
|
|
267
|
+
current = max(downloaded, disk_high_water, self._hub_bytes())
|
|
268
|
+
if current > downloaded:
|
|
269
|
+
downloaded = current
|
|
270
|
+
last_change_at = now
|
|
271
|
+
elapsed = now - last_emit_at
|
|
272
|
+
if downloaded == 0 or elapsed < EMIT_INTERVAL_SECONDS:
|
|
273
|
+
continue
|
|
274
|
+
growth = downloaded - emitted
|
|
275
|
+
emitted = downloaded
|
|
276
|
+
last_emit_at = now
|
|
277
|
+
# Flat bytes still report, so phase_detail stops showing the
|
|
278
|
+
# last healthy rate - but do not count as progress, so the
|
|
279
|
+
# stall watchdog sees the silence a hang is
|
|
280
|
+
self._context.emit(
|
|
281
|
+
"download_progress",
|
|
282
|
+
counts_as_progress=growth > 0,
|
|
283
|
+
repo_id=self._repo_id,
|
|
284
|
+
downloaded_bytes=downloaded,
|
|
285
|
+
bytes_per_second=growth / elapsed,
|
|
286
|
+
seconds_since_bytes_changed=round(now - last_change_at, 1),
|
|
287
|
+
)
|
|
288
|
+
except Exception as e:
|
|
289
|
+
# A watch that breaks must not take the download down with it
|
|
290
|
+
logger.debug(f"Download watch for '{self._repo_id}' failed: {e}")
|
|
291
|
+
return
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _format_bytes(n):
|
|
295
|
+
# Decimal units, as the labels say and as the hub itself reports sizes -
|
|
296
|
+
# 1,879,623,333 bytes is 1.9 GB (it is 1.75 GiB)
|
|
297
|
+
for unit in ("B", "KB", "MB", "GB", "TB"):
|
|
298
|
+
if n < 1000 or unit == "TB":
|
|
299
|
+
return f"{n:.1f} {unit}" if unit != "B" else f"{n:.0f} {unit}"
|
|
300
|
+
n /= 1000
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def format_progress(
|
|
304
|
+
repo_id, downloaded_bytes, bytes_per_second, seconds_since_bytes_changed=None
|
|
305
|
+
):
|
|
306
|
+
"""The `phase_detail` text a job's progress reports while this repo is
|
|
307
|
+
downloading - `downloading <repo>: 12.3 GB, 38.0 MB/s` per #343, or
|
|
308
|
+
`downloading <repo>: 12.3 GB, no bytes for 45s` once bytes stop."""
|
|
309
|
+
text = f"downloading {repo_id}: {_format_bytes(downloaded_bytes or 0)}"
|
|
310
|
+
if bytes_per_second:
|
|
311
|
+
text += f", {_format_bytes(bytes_per_second)}/s"
|
|
312
|
+
elif seconds_since_bytes_changed:
|
|
313
|
+
text += f", no bytes for {seconds_since_bytes_changed:.0f}s"
|
|
314
|
+
return text
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def watch(repo_id, cache_dir=None, repo_type="model"):
|
|
318
|
+
"""A DownloadWatch on the active run context, or a no-op context manager
|
|
319
|
+
when repo_id is not shaped like a hub repo (a local path, for instance)."""
|
|
320
|
+
if not is_watchable_repo_id(repo_id):
|
|
321
|
+
return _NULL_WATCH
|
|
322
|
+
return DownloadWatch(
|
|
323
|
+
repo_id, get_context(), cache_dir=cache_dir, repo_type=repo_type
|
|
324
|
+
)
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
class _NullWatch:
|
|
328
|
+
def __enter__(self):
|
|
329
|
+
return self
|
|
330
|
+
|
|
331
|
+
def __exit__(self, *exc_info):
|
|
332
|
+
return False
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
_NULL_WATCH = _NullWatch()
|