cozyclay 1.9.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (286) hide show
  1. package/CHANGELOG.md +171 -0
  2. package/bin/agent/agent-routes.mjs +74 -70
  3. package/bin/agent/agent-runner.mjs +10 -3
  4. package/bin/agent/motion-runtime.mjs +32 -260
  5. package/bin/agent/providers.mjs +46 -3
  6. package/bin/agent/session-store.mjs +41 -5
  7. package/bin/agent/studio-prompt.mjs +1 -1
  8. package/bin/agent/studio-tools.mjs +48 -11
  9. package/bin/cozyclay.mjs +26 -8
  10. package/bin/live/cli.mjs +22 -3
  11. package/bin/mcp-runtime.mjs +35 -20
  12. package/bin/telemetry-state.mjs +5 -2
  13. package/bin/update-check.mjs +15 -0
  14. package/dist/app/index.html +6 -6
  15. package/dist/assets/analytics-MVZOEW8O.js +1 -0
  16. package/dist/assets/app-C1sdVbg4.css +1 -0
  17. package/dist/assets/app-CA55tcI3.js +4876 -0
  18. package/dist/assets/{demo-B-eg1CT8.js → demo-ByeYp-dY.js} +1 -1
  19. package/dist/assets/{first-shot-handoff-CGgnvNu8.js → first-shot-handoff-DgT28ZiV.js} +1 -1
  20. package/dist/assets/{landing-IbRp3FnQ.js → landing-BaXkpFip.js} +1 -1
  21. package/dist/assets/shot-prompt-CbqWcWYi.css +1 -0
  22. package/dist/assets/shot-prompt-Yrt99wZx.js +17 -0
  23. package/dist/assets/{ticket-BgIzOBOf.js → ticket-BcmsnHn5.js} +1 -1
  24. package/dist/assets/workflow-DRdwoDBT.css +1 -0
  25. package/dist/assets/workflow-dNPGY9xz.js +179 -0
  26. package/dist/cozyclay-package.json +1 -1
  27. package/dist/fonts/IBM-Plex-Sans-OFL.txt +92 -0
  28. package/dist/fonts/JetBrains-Mono-OFL.txt +93 -0
  29. package/dist/fonts/README.md +13 -9
  30. package/dist/fonts/ibm-plex-sans-400-latin.woff2 +0 -0
  31. package/dist/fonts/ibm-plex-sans-500-latin.woff2 +0 -0
  32. package/dist/fonts/ibm-plex-sans-600-latin.woff2 +0 -0
  33. package/dist/fonts/jetbrains-mono-400-latin.woff2 +0 -0
  34. package/dist/fonts/jetbrains-mono-500-latin.woff2 +0 -0
  35. package/dist/index.html +7 -7
  36. package/dist/privacy/index.html +13 -8
  37. package/dist/sitemap.xml +6 -6
  38. package/dist/workflow/index.html +5 -5
  39. package/mcp/LIVE-PROTOCOL.md +1 -1
  40. package/mcp/live-hub.mjs +35 -93
  41. package/mcp/mesh-file.mjs +48 -0
  42. package/mcp/server.mjs +6 -77
  43. package/mcp/tool-handlers.mjs +224 -377
  44. package/package.json +2 -1
  45. package/src/App.jsx +1273 -9075
  46. package/src/analytics.js +393 -9
  47. package/src/app-context.js +211 -0
  48. package/src/app-stage.jsx +25 -24
  49. package/src/ardy/auto-fix-panel.css +108 -0
  50. package/src/ardy/collision-blockers.js +9 -3
  51. package/src/ardy/fix-collisions.js +10 -66
  52. package/src/ardy/ground.js +4 -4
  53. package/src/ardy/ik-drag.js +101 -0
  54. package/src/ardy/ik-key-json.js +43 -0
  55. package/src/ardy/ik.js +122 -76
  56. package/src/ardy/physics-panel.css +2 -0
  57. package/src/ardy/physics-panel.jsx +17 -5
  58. package/src/ardy/platform-fit-panel.css +64 -0
  59. package/src/ardy/platform-fit-panel.jsx +40 -0
  60. package/src/ardy/platform-fit.js +326 -0
  61. package/src/ardy/playback.js +166 -20
  62. package/src/ardy/range-pin.js +299 -0
  63. package/src/ardy/timeline-coordinates.js +26 -0
  64. package/src/ardy/timeline.css +1199 -0
  65. package/src/ardy/timeline.jsx +149 -41
  66. package/src/ardy/waypoints.js +2 -2
  67. package/src/asset-pane.css +462 -0
  68. package/src/asset-pane.jsx +461 -232
  69. package/src/command-bus.js +349 -0
  70. package/src/commands/ai.js +31 -0
  71. package/src/commands/cast.js +140 -0
  72. package/src/commands/elements/character.js +13 -0
  73. package/src/commands/elements/motion.js +14 -0
  74. package/src/commands/elements/object.js +22 -0
  75. package/src/commands/elements/scene.js +16 -0
  76. package/src/commands/elements/shot.js +9 -0
  77. package/src/commands/elements/stage.js +11 -0
  78. package/src/commands/elements.js +155 -0
  79. package/src/commands/export.js +20 -0
  80. package/src/commands/index.js +40 -0
  81. package/src/commands/motion.js +231 -0
  82. package/src/commands/objects.js +174 -0
  83. package/src/commands/project.js +53 -0
  84. package/src/commands/scene.js +81 -0
  85. package/src/commands/shared.js +19 -0
  86. package/src/commands/shot.js +103 -0
  87. package/src/commands/stage.js +17 -0
  88. package/src/commands/view.js +42 -0
  89. package/src/document-store.js +144 -0
  90. package/src/domains/cast.js +1007 -0
  91. package/src/domains/motion.js +3219 -0
  92. package/src/domains/objects.js +932 -0
  93. package/src/domains/scenes.js +817 -0
  94. package/src/domains/shots.js +404 -0
  95. package/src/domains/stage.js +93 -0
  96. package/src/dualview.jsx +113 -57
  97. package/src/facing-marks.js +3 -0
  98. package/src/first-success-guide.jsx +14 -14
  99. package/src/grid-view.js +18 -5
  100. package/src/hierarchy-model.js +24 -16
  101. package/src/hierarchy-panel.css +386 -0
  102. package/src/hierarchy-panel.jsx +150 -21
  103. package/src/{fal-motion-client.js → i2v-motion-client.js} +13 -13
  104. package/src/i2v-motion-studio.jsx +180 -0
  105. package/src/ik-camera.js +3 -0
  106. package/src/main.jsx +6 -0
  107. package/src/motion/generation.js +105 -0
  108. package/src/motion-readiness-ui.jsx +4 -4
  109. package/src/motion-readiness.js +11 -0
  110. package/src/motion-trail.js +366 -9
  111. package/src/object-gizmo.jsx +7 -1
  112. package/src/otio.js +2 -1
  113. package/src/panels/CameraPanel.jsx +60 -0
  114. package/src/panels/CharacterTransformPanel.jsx +48 -0
  115. package/src/panels/EnvironmentPanel.jsx +40 -0
  116. package/src/panels/Foldout.jsx +32 -0
  117. package/src/panels/LightPanel.jsx +24 -0
  118. package/src/panels/ObjectTransformPanel.jsx +554 -0
  119. package/src/panels/PosePanel.jsx +97 -0
  120. package/src/panels/ProjectPanel.jsx +33 -0
  121. package/src/panels/PromptBlocksPanel.jsx +442 -0
  122. package/src/panels/PropsPanel.jsx +70 -0
  123. package/src/panels/ReferenceImageField.jsx +91 -0
  124. package/src/panels/RigControlPanel.jsx +78 -0
  125. package/src/panels/RigPanel.jsx +37 -0
  126. package/src/panels/SubjectBox.jsx +47 -0
  127. package/src/panels/SubjectsPanel.jsx +39 -0
  128. package/src/panels/VideoCapturePanel.jsx +180 -0
  129. package/src/panels/details.css +1171 -0
  130. package/src/panels/motion.css +197 -0
  131. package/src/panels/pose.css +39 -0
  132. package/src/planview.jsx +60 -31
  133. package/src/posestudio.jsx +59 -4
  134. package/src/project-browser.css +1073 -0
  135. package/src/project-browser.jsx +284 -123
  136. package/src/range-pin-object-transform.js +19 -0
  137. package/src/range-pin-panel.css +518 -0
  138. package/src/range-pin-panel.jsx +344 -0
  139. package/src/result-modal.jsx +5 -5
  140. package/src/room.jsx +79 -51
  141. package/src/scene-objects.js +37 -1
  142. package/src/scenes.js +5 -0
  143. package/src/semantic-edit.js +2 -0
  144. package/src/settings-menu.jsx +39 -105
  145. package/src/shell/BottomDock.jsx +450 -0
  146. package/src/shell/DetailsSlot.jsx +513 -0
  147. package/src/shell/LibrarySlot.jsx +86 -0
  148. package/src/shell/MenuBar.jsx +601 -0
  149. package/src/shell/OutlinerSlot.jsx +50 -0
  150. package/src/shell/PreferencesDialog.jsx +538 -0
  151. package/src/shell/PreferencesSlot.jsx +40 -0
  152. package/src/shell/StatusBar.jsx +125 -0
  153. package/src/shell/StudioShell.jsx +92 -0
  154. package/src/shell/TopBar.jsx +129 -0
  155. package/src/shell/ViewportToolbar.jsx +656 -0
  156. package/src/shell/agent-glass.css +296 -0
  157. package/src/shell/dock.css +93 -0
  158. package/src/shell/glass-regions.css +985 -0
  159. package/src/shell/glass.css +687 -0
  160. package/src/shell/log-store.js +52 -0
  161. package/src/shell/mode.css +9 -0
  162. package/src/shell/preferences.css +651 -0
  163. package/src/shell/shell.css +282 -0
  164. package/src/shell/studio-shell-context.js +19 -0
  165. package/src/shell/topbar.css +435 -0
  166. package/src/shell/viewport.css +451 -0
  167. package/src/store/authored-intent.js +22 -0
  168. package/src/store/runtime-adapters.js +92 -0
  169. package/src/store/scene-stage.js +21 -0
  170. package/src/store/use-document-store.js +14 -0
  171. package/src/studio-actions.js +210 -0
  172. package/src/studio-agent-commands.js +41 -271
  173. package/src/studio-agent-context.js +39 -10
  174. package/src/studio-agent-motion.js +122 -373
  175. package/src/studio-agent-protocol.js +67 -31
  176. package/src/studio-app-binding.js +379 -0
  177. package/src/studio-contact-sheet.js +73 -0
  178. package/src/studio-elements.js +58 -51
  179. package/src/styles/themes.css +200 -0
  180. package/src/styles/tokens.css +69 -0
  181. package/src/styles.css +460 -1742
  182. package/src/theme.js +44 -0
  183. package/src/timeline-extent.js +16 -0
  184. package/src/trail-key-conflicts.js +39 -0
  185. package/src/trail-pick.js +59 -0
  186. package/src/ui.jsx +8 -8
  187. package/src/use-case-question.jsx +66 -0
  188. package/src/workflow/AgentPanel.jsx +82 -69
  189. package/src/workflow/agent-client.js +12 -2
  190. package/src/workflow/agent-panel.css +34 -32
  191. package/src/workflow/cozy-scene-node.css +2 -0
  192. package/src/workflow/workflow.css +2 -0
  193. package/tools/ardy/__pycache__/cclay_gvhmr_worker.cpython-313.pyc +0 -0
  194. package/tools/ardy/bridge.mjs +164 -146
  195. package/tools/ardy/visual-qa.mjs +5 -5
  196. package/tools/bench/EXP3.md +61 -0
  197. package/tools/bench/cclay_bench_extract_incam.py +125 -0
  198. package/tools/bench/cclay_bench_extract_obs.py +204 -0
  199. package/tools/bench/cclay_bench_runner.py +42 -0
  200. package/tools/bench/cube-contact.mjs +162 -0
  201. package/tools/bench/exp3.mjs +152 -0
  202. package/tools/bench/extract-bench-lib.mjs +216 -0
  203. package/tools/bench/extract-bench.mjs +302 -0
  204. package/tools/bench/fal-generate.mjs +88 -0
  205. package/tools/bench/fit/README.md +189 -0
  206. package/tools/bench/fit/camera.mjs +59 -0
  207. package/tools/bench/fit/contact.mjs +419 -0
  208. package/tools/bench/fit/footlock.mjs +234 -0
  209. package/tools/bench/fit/motion.mjs +85 -0
  210. package/tools/bench/fit/pin.mjs +27 -0
  211. package/tools/bench/fit/remote.mjs +105 -0
  212. package/tools/bench/fit-bench.mjs +126 -0
  213. package/tools/bench/fit-sanity.mjs +45 -0
  214. package/tools/bench/metrics.mjs +313 -0
  215. package/tools/bench/obs/depth.mjs +258 -0
  216. package/tools/bench/obs/extrinsics.mjs +175 -0
  217. package/tools/bench/obs/fit_mannequin_betas.py +340 -0
  218. package/tools/bench/obs/ground.mjs +294 -0
  219. package/tools/bench/obs/heading.mjs +151 -0
  220. package/tools/bench/obs/ladder.mjs +555 -0
  221. package/tools/bench/obs/mannequin-betas.json +124 -0
  222. package/tools/bench/obs/remote.mjs +83 -0
  223. package/tools/bench/obs/rest_joints.py +52 -0
  224. package/tools/bench/obs/ybot-targets.mjs +49 -0
  225. package/tools/bench/obs-bench.mjs +520 -0
  226. package/tools/bench/score.mjs +343 -0
  227. package/tools/bench/summarize.mjs +86 -0
  228. package/tools/dev/pages/privacy.html +13 -8
  229. package/tools/gt-render/browser.mjs +159 -0
  230. package/tools/gt-render/camera-math.mjs +276 -0
  231. package/tools/gt-render/page.mjs +267 -0
  232. package/tools/gt-render/render.mjs +514 -0
  233. package/tools/gt-render/scene-box.mjs +31 -0
  234. package/tools/gt-render/take-transform.mjs +87 -0
  235. package/tools/morphgs/assets/gen_truth.py +39 -0
  236. package/tools/morphgs/assets/mesh_ori_rig.txt +27 -0
  237. package/tools/morphgs/assets/playback-check.mjs +30 -0
  238. package/tools/morphgs/demo-gate.sh +26 -0
  239. package/tools/morphgs/fbx2morphgs.mjs +68 -0
  240. package/tools/morphgs/morphgs-to-cskel27.mjs +105 -0
  241. package/tools/morphgs/patches/preprocess_src-none-mode.patch +76 -0
  242. package/tools/morphgs/setup-on-cluster.sh +126 -0
  243. package/tools/qa/css-rule-usage.mjs +283 -0
  244. package/tools/qa/studio-control-count.mjs +70 -32
  245. package/tools/run-tests.mjs +134 -12
  246. package/tools/track/DESIGN.md +513 -0
  247. package/tools/track/backfill-provenance.mjs +116 -0
  248. package/tools/track/budget.mjs +8 -0
  249. package/tools/track/check-rig.mjs +209 -0
  250. package/tools/track/diagnostics.schema.json +60 -0
  251. package/tools/track/export-rig.mjs +256 -0
  252. package/tools/track/fallback.mjs +7 -0
  253. package/tools/track/fk-parity-fixture.mjs +161 -0
  254. package/tools/track/gate.mjs +349 -0
  255. package/tools/track/masks.mjs +318 -0
  256. package/tools/track/metrics.mjs +201 -0
  257. package/tools/track/publish-obs.mjs +108 -0
  258. package/tools/track/py/check_env.py +61 -0
  259. package/tools/track/py/eval_lr.py +229 -0
  260. package/tools/track/py/lr_viterbi.py +294 -0
  261. package/tools/track/py/masks.py +405 -0
  262. package/tools/track/py/objective.py +382 -0
  263. package/tools/track/py/rig.py +508 -0
  264. package/tools/track/py/scene.py +326 -0
  265. package/tools/track/py/test_joint_indices.py +123 -0
  266. package/tools/track/py/test_lr_viterbi.py +195 -0
  267. package/tools/track/py/test_masks.py +229 -0
  268. package/tools/track/py/test_rig.py +288 -0
  269. package/tools/track/py/test_scene.py +354 -0
  270. package/tools/track/py/test_track.py +328 -0
  271. package/tools/track/py/track.py +411 -0
  272. package/tools/track/remote.mjs +140 -0
  273. package/tools/track/rig-dump.mjs +227 -0
  274. package/tools/track/run-box-tests.mjs +14 -0
  275. package/tools/track/run-box.mjs +168 -0
  276. package/tools/track/setup-box.sh +59 -0
  277. package/tools/track/study-2d.mjs +728 -0
  278. package/dist/assets/analytics-B1hnH66c.js +0 -1
  279. package/dist/assets/app-BWusbqgO.js +0 -4861
  280. package/dist/assets/app-qKDo4PBX.css +0 -1
  281. package/dist/assets/shot-prompt-CHUtw6af.js +0 -17
  282. package/dist/assets/shot-prompt-C_g2BHVc.css +0 -1
  283. package/dist/assets/workflow-CVwzMhz2.js +0 -179
  284. package/dist/assets/workflow-DsudxKHF.css +0 -1
  285. package/src/fal-motion-studio.jsx +0 -180
  286. package/src/scene-history.js +0 -129
@@ -0,0 +1,513 @@
1
+ # Model-based monocular tracker for cskel27
2
+
3
+ ## Implemented v1 contract (takes precedence over the historical proposal below)
4
+
5
+ The mocap-rearch execution plan intentionally supersedes the proposal's renderer,
6
+ learned prior, optical flow, mask re-prompting, and LBFGS suggestions. V1 uses
7
+ only robust joint reprojection, exterior signed-DT surface sampling plus a
8
+ foreground coverage Chamfer, ray/OBB visibility, acceleration, rest-relative
9
+ angle caps, stance skate, and mesh/floor/box non-penetration. It never loads
10
+ reference motion or reference image assets. The original proposal is preserved
11
+ below for provenance, not as authorization for its deferred features.
12
+
13
+ `track.py` consumes the seven files declared by `remote.mjs`. Calibration and
14
+ optional endpoint poses travel inside `camera.json` as
15
+ `tracker: {deltaPx, cameraFixed, endpoints: {a?, b?}}`. Each endpoint contains
16
+ `rotMats` (27x3x3 or 243 flat local rotation values) and `rootPos` (3 floats),
17
+ already in the initializer's coordinate frame. Defaults are the measured Gate-0
18
+ 5.708 px and fixed camera. When fixed, **all** nuisance values are frozen
19
+ (scale 1, yaw/pitch/focal deltas zero). Otherwise their bounds and Gaussian
20
+ priors are those in todo 10. No root anchoring is added or removed.
21
+
22
+ Gate-2 round 1 keeps 150 root-only Adam steps at 0.01 and 300 full steps at
23
+ 0.005, then uses 200 full steps at 0.001 with triple penetration weight.
24
+ Silhouette weight remains 6.0 and keypoint weight 1. Reducing pose/refine
25
+ silhouette to 1.25 failed seed-102 synthetic acceptance even without the added
26
+ temporal term (21.65 mm root error), while the controlled 6.0 candidate
27
+ recovered 17.40 mm. The round therefore retains depth support and uses the
28
+ bounded initializer prior below to discourage unsupported articulation.
29
+
30
+ A bounded G5 trust-region prior is off during root initialization, weighted
31
+ 0.1 during pose, and linearly annealed from 0.1 to zero during refine. The
32
+ squared root-path deviation is centred within each optimization window (free
33
+ rigid placement correction), normalized by 25 cm, and bounded as d2/(1+d2).
34
+ Rotation uses squared chord distance normalized to a 30-degree rotation, with
35
+ the same bound. Root-path weight is 2; root-rotation weight is 2; shoulder,
36
+ upper-arm, forearm and hand rotation weights are 4; other joints are 1.
37
+ Only the supplied initializer is used, never truth or a learned prior.
38
+
39
+ The existing 0.05 whole-body acceleration term remains unchanged. An added
40
+ 0.075 proximal acceleration term was rejected after a controlled synthetic
41
+ comparison isolated a depth bias: removing only that term reduced seed-102
42
+ root error from 21.34 to 17.40 mm; removing trust or restoring the old learning
43
+ rate did not fix it. Interim silhouette increases and translation-only polish
44
+ were also discarded. No acceptance threshold was changed.
45
+ Assignments and stance are re-estimated at block boundaries. Clips through 240
46
+ frames are whole batches; longer clips use 120-frame windows with 16-frame
47
+ linear overlaps. Each coverage query participates; 64-query nearest-neighbour
48
+ tiles bound memory instead of dropping a loss. Surface samples are fixed
49
+ area-stratified barycentric samples of the decimated Studio mesh.
50
+
51
+ Signed DT values are clamped at zero for *interior* samples: pushing every
52
+ surface sample to a silhouette boundary would hollow out the body. The reverse
53
+ foreground Chamfer supplies coverage. Rejection IoU is a named half-resolution
54
+ point-splat occupancy estimate (`sampleMaskIoUMean`), not a rasterizer or the
55
+ independent benchmark score. Benchmark scoring remains the quality authority.
56
+
57
+ Keypoint confidence is sanitized at the input boundary, not rejected: ViTPose
58
+ confidence is the raw flip-averaged heatmap peak of a height-1 Gaussian
59
+ regression (GVHMR `VitPoseExtractor` -> mmpose `keypoints_from_heatmaps`
60
+ `maxvals`), not a probability, and overshoots 1 (up to 1.0385 on the Gate-2 r0
61
+ obs cache). Non-finite confidence becomes 0, then values are clipped to [0,1].
62
+ The counts are reported as `stageLosses` entries `input.confidenceNonFinite`,
63
+ `input.confidenceAboveOne` and `input.confidenceBelowZero`. Keypoint
64
+ coordinates, the kp2d shape and every other input remain strictly validated
65
+ (exit 2).
66
+
67
+ `--track-ablate` selects cumulative fit components for the Gate-2 ablation
68
+ table; weights and schedule are identical at every level, and keypoints plus
69
+ priors (acceleration, limits, endpoints, nuisance) are always on:
70
+
71
+ | level | adds |
72
+ |---|---|
73
+ | `kp-only` | keypoint reprojection with detector left/right labels as given |
74
+ | `silhouette` | signed-DT boundary + coverage Chamfer |
75
+ | `viterbi` | latent L/R Viterbi assignment |
76
+ | `contacts` | stance skate + floor/box penetration |
77
+ | `full` (default) | ray/box occlusion of keypoints and surface samples in the fit |
78
+
79
+ Disabled terms are not evaluated and are absent from `stageLosses`; the level
80
+ is recorded as `stageLosses["ablation.<level>"] = 1`. Diagnostics (occlusion
81
+ flags, stance, penetration, sample IoU, visible keypoint residual) and the
82
+ rejection rules are computed identically at every level, so an ablated fit is
83
+ rejected by the same policy as the full one. Without Viterbi, `lrState` is
84
+ `identity` with zero margin and not ambiguous. obs-bench forwards
85
+ `--track-ablate <level>` through `runTracker({ablate})`.
86
+
87
+ Exit 3 returns diagnostics and no motion for no evidence, non-finite fits,
88
+ resource exhaustion, excessive visible reprojection error or mask mismatch.
89
+ Exit 2 is malformed input. A zero-iteration request preserves every initializer
90
+ motion member exactly (fps comes from obs) and bypasses fitted-residual rejection;
91
+ it exists to test the coordinate/output contract, not claim a successful fit.
92
+
93
+ ---
94
+
95
+ ## Design decision
96
+
97
+ Use the current GVHMR result only as an initializer and proposal for image evidence. Fit a single physically constrained trajectory for the known cskel27 character over the entire clip, using the known camera, mesh, floor, and boxes in the objective. The optimizer should be implemented in PyTorch/CUDA with a differentiable rasterizer and should alternate continuous optimization with two small discrete dynamic programs:
98
+
99
+ 1. a Viterbi pass for latent left/right keypoint assignments; and
100
+ 2. a stance/contact pass for each foot.
101
+
102
+ This directly addresses the observed failures: ViTPose left/right swaps causing 97-178 degree yaw flips, box-occluded legs producing wrong depth, and root-only corrections causing foot skate (`context.md`, lines 16-17). It does not use palette or color cues; the palette approach is explicitly abandoned (`context.md`, lines 19-21).
103
+
104
+ The fit should retain per-frame uncertainty and a failure flag. It must not turn an unobserved body part behind a box into a high-confidence measurement.
105
+
106
+ ## 1. State and parametrization
107
+
108
+ Let frames be `t = 0..T-1`, with image size `W x H`, known camera intrinsics `K`, and fixed world-to-camera extrinsics `(R_cw, t_cw)`. Use the Studio/world coordinate system for all scene constraints.
109
+
110
+ For every frame optimize
111
+
112
+ ```
113
+ q_t = (x_t, a_t, u_{t,0}, ..., u_{t,J-1})
114
+ ```
115
+
116
+ where:
117
+
118
+ - `x_t in R^3` is the root/pelvis translation in world coordinates;
119
+ - `a_t in R^6` is the root rotation's continuous 6D representation;
120
+ - `u_{t,j} in R^6` is the local rotation representation for joint `j`.
121
+
122
+ For a 6D vector `a = [a1,a2]`, convert to a rotation matrix by
123
+
124
+ ```
125
+ b1 = normalize(a1)
126
+ b2 = normalize(a2 - b1 * dot(b1,a2))
127
+ b3 = cross(b1,b2)
128
+ R6D(a) = [b1 b2 b3]
129
+ ```
130
+
131
+ with columns `b1,b2,b3`. In code, normalize with a small fixed epsilon only for numerical stability; do not use a second, unconstrained rotation representation in the state.
132
+
133
+ Use root rotation `R_t = R6D(a_t)`. For each joint, either optimize a local delta from the rig rest pose,
134
+
135
+ ```
136
+ R_{t,j}^{local} = R^{rest}_j R6D(u_{t,j}),
137
+ ```
138
+
139
+ or, if the existing cskel27 convention already defines zero as the rest pose, use `R6D(u_{t,j})` directly. Pick one convention and test it on a rest-pose identity case. The preferred interface is a local delta because joint-limit ranges are then stable across clips.
140
+
141
+ Forward kinematics gives every joint and bone transform:
142
+
143
+ ```
144
+ G_{t,root} = [R_t, x_t]
145
+ G_{t,j} = G_{t,parent(j)} [R_{t,j}^{local}, l_j]
146
+ X_{t,j} = translation(G_{t,j}),
147
+ ```
148
+
149
+ where `l_j` is the known rest offset. The skinned mesh is
150
+
151
+ ```
152
+ V_t(v) = sum_k w_{v,k} (G_{t,k} G^{-1}_{rest,k}) V_rest(v),
153
+ ```
154
+
155
+ using the existing known weights and mesh. Do not infer scale. The known body size and camera remove the scale ambiguity that currently remains in a generic regressor (`context.md`, lines 3-5 and 9).
156
+
157
+ The optimizer's primary variables can be packed as `[x, a, u]` in float32. Use a separate float32 latent velocity/contact state only if needed; do not optimize camera, bone lengths, or mesh shape.
158
+
159
+ ### Projection and visibility interface
160
+
161
+ For any world point `X`,
162
+
163
+ ```
164
+ Xc = R_cw X + t_cw
165
+ pi(X) = (K Xc)_{xy} / (K Xc)_z.
166
+ ```
167
+
168
+ Reject points with `Xc.z <= 0`. The renderer receives `V_t`, the known box meshes, the floor, `K`, and extrinsics, and returns:
169
+
170
+ - soft character silhouette `S_t^q(u) in [0,1]`;
171
+ - character depth `Z_t^q(u)`;
172
+ - scene-occluded character silhouette `S_t^{q,visible}`;
173
+ - per-vertex/per-joint visibility `v_{t,i}`;
174
+ - optionally a differentiable triangle ID or barycentric map.
175
+
176
+ The scene-aware visibility output is important: a leg behind a known box should be excluded from image evidence, not forced to the front of the box.
177
+
178
+ ## 2. Data terms
179
+
180
+ ### 2.1 Keypoint reprojection
181
+
182
+ Let the 2D detector provide candidates `y_{t,k} in R^2`, confidence `c_{t,k} in [0,1]`, and detector covariance `Sigma_{t,k}` where available. Do not assume the detector's L/R labels are correct. For anatomical joint `i`, let `p_{t,i}(q) = pi(X_{t,i}(q))`.
183
+
184
+ For a fixed assignment `h_t` of detector labels to anatomical labels, use a confidence-weighted robust Mahalanobis residual:
185
+
186
+ ```
187
+ r^kp_{t,i}(q,h_t) =
188
+ sqrt(c_{t,h_t(i)} * v_{t,i}) *
189
+ rho_delta( || L_{t,h_t(i)} (p_{t,i}(q)-y_{t,h_t(i)}) ||_2 ),
190
+ ```
191
+
192
+ where `L L^T = Sigma^{-1}`; use an isotropic detector scale when covariance is unavailable. `rho_delta` is Huber or Geman-McClure. Huber is easier to optimize:
193
+
194
+ ```
195
+ rho_delta(r) = 0.5 r^2 if r <= delta
196
+ delta*(r - 0.5*delta) otherwise.
197
+ ```
198
+
199
+ Set `v_{t,i}` to zero or a small floor when the renderer says the joint is behind a box. A detector confidence alone must not revive an occluded observation.
200
+
201
+ Include a pelvis/root observation only if its detector confidence is valid. The known camera and mesh mean that a stable silhouette and known bone lengths can constrain depth even when a particular 2D joint is absent.
202
+
203
+ ### 2.2 Latent L/R assignment without color
204
+
205
+ Use a small permitted permutation set `H`. At minimum it contains identity and a global L/R exchange. For robustness to independent arm/leg swaps, use the product of pairwise swaps for the bilateral groups `{shoulder, elbow, wrist}`, `{hip, knee, ankle}`, and any available foot/toe groups, yielding at most `2^B` states. Do not allow arbitrary permutations: they would explain detector noise by anatomically nonsensical relabeling.
206
+
207
+ At the current continuous trajectory `q`, define the emission cost for state `h`:
208
+
209
+ ```
210
+ D_t(h) = sum_i c_{t,h(i)} v_{t,i}
211
+ rho_delta( ||p_{t,i}(q)-y_{t,h(i)}|| / s_i )
212
+ + lambda_h * complexity(h).
213
+ ```
214
+
215
+ Here `s_i` is a pixel scale proportional to torso height, making the cost resolution-independent. Define an assignment-aware transition cost using expected image motion and a switch penalty:
216
+
217
+ ```
218
+ T_t(h',h) = lambda_switch * Hamming(h',h)
219
+ + lambda_cont * sum_i w_i rho_delta(
220
+ || (y_{t,h(i)} - y_{t-1,h'(i)})
221
+ - (p_{t,i}(q)-p_{t-1,i}(q)) || / s_i ).
222
+ ```
223
+
224
+ The Viterbi recurrence is
225
+
226
+ ```
227
+ C_0(h) = D_0(h)
228
+ C_t(h) = D_t(h) + min_{h'} [C_{t-1}(h') + T_t(h',h)].
229
+ ```
230
+
231
+ Store backpointers and recover `h_0:T-1`. Run this after each few continuous optimization epochs, not once permanently at initialization. Freeze the selected path for the next continuous block. Add hysteresis: only change a frame's assignment if the new Viterbi path improves the normalized cost by a margin, for example `0.1` robust residual units per active bilateral joint. This prevents assignment chatter.
232
+
233
+ This is a temporal inference problem, not a color-classification problem. The transition term makes a one-frame swap expensive while still allowing a sustained true relabeling. Report the Viterbi margin between the best and second-best path as an ambiguity diagnostic.
234
+
235
+ ### 2.3 Silhouette/mask term
236
+
237
+ Use SAM2's video predictor as the primary segmenter, initialized in the first frame with a person box/positive points from the projected GVHMR mesh and negative points outside it. Propagate the mask through the clip, then optionally correct selected frames with the current rendered silhouette as a new prompt. SAM2 is preferable to independent per-frame segmentation because the camera is fixed and temporal consistency matters; it also avoids relying on the mannequin's gray color.
238
+
239
+ Store the SAM2 mask probability/logit, not just a hard binary mask. Remove obvious background/box pixels using the known scene render where possible. Let `M_t(u) in [0,1]` be the observed character mask probability and `S_t^{q,visible}(u)` the differentiably rendered visible character mask.
240
+
241
+ Use both a region and boundary term:
242
+
243
+ ```
244
+ E_mask(t) = lambda_bce * mean_u BCE(S_t^{q,visible}(u), M_t(u))
245
+ + lambda_dtm * mean_{u in band(M_t)}
246
+ rho_delta( |DT(M_t)(u) - DT(S_t^{q,visible})(u)| / tau )
247
+ + lambda_area * rho_delta(area(S_t^{q,visible})/area(M_t)-1).
248
+ ```
249
+
250
+ In practice, calculate signed distance transforms from detached masks for the boundary term, or use a soft rasterizer plus a narrow-band Chamfer loss. Downweight pixels that the known boxes hide and pixels with low SAM2 confidence. Use an image pyramid so coarse silhouette alignment moves the root/depth before fine boundary optimization.
251
+
252
+ A single global mask term is not enough to recover articulation, but it is especially valuable for the current box-occlusion cases where keypoints disappear.
253
+
254
+ ### 2.4 Optional dense motion term
255
+
256
+ Add dense optical flow only when its confidence and forward/backward consistency are good. Use RAFT or GMFlow on consecutive frames. For a visible mesh sample `v` at time `t`, with projected coordinate `u_{t,v}`, define
257
+
258
+ ```
259
+ E_flow = sum_{t,v} w_flow(t,v) rho_delta(
260
+ || [u_{t+1,v}(q)-u_{t,v}(q)] - F_t(u_{t,v}(q)) || / s_flow
261
+ ).
262
+ ```
263
+
264
+ Reject samples near occlusion boundaries, with invalid depth ordering, or with inconsistent forward/backward flow. Compare the rendered point only to flow in the foreground region. The weight should be substantially lower than keypoints and silhouette because generated video can violate brightness constancy and a textureless clay mesh offers weak correspondences. A feature correspondence variant can use DINOv2/LoFTR descriptors, but it must be optional: on a faceless, nearly uniform mannequin, learned feature matches can be less reliable than the silhouette.
265
+
266
+ The complete data term is
267
+
268
+ ```
269
+ E_data = E_kp + E_mask + E_flow.
270
+ ```
271
+
272
+ Normalize each term by the number of valid observations so a long visible torso does not overwhelm all joints.
273
+
274
+ ## 3. Prior terms
275
+
276
+ ### 3.1 Temporal acceleration
277
+
278
+ For root translation use second differences in world space:
279
+
280
+ ```
281
+ E_acc_root = sum_{t=1}^{T-2} rho(||x_{t+1} - 2*x_t + x_{t-1}|| / s_x).
282
+ ```
283
+
284
+ For each joint use the Lie-algebra relative rotation
285
+
286
+ ```
287
+ omega_{t,j} = log( R_{t,j}^{local} ) in R^3
288
+ ```
289
+
290
+ and penalize angular acceleration using relative increments rather than subtracting matrices:
291
+
292
+ ```
293
+ delta_{t,j} = log( R_{t,j}^{local}(t-1)^T R_{t,j}^{local}(t) )
294
+ E_acc_rot = sum_{t,j} rho(||delta_{t+1,j}-delta_{t,j}|| / s_theta).
295
+ ```
296
+
297
+ Also add a lower-weight first-difference term only for detector dropout or severe ambiguity. A first-difference penalty everywhere would reproduce the current over-smoothed result; the current bench already reports limb acceleration below truth (`context.md`, lines 13-15). Robust acceleration is preferable to a large quadratic smoothness weight so a genuine bump is retained.
298
+
299
+ ### 3.2 Learned motion prior
300
+
301
+ Use a weak prior trained or selected on retargeted cskel27/AMASS-like motion, not a prior that changes the rig geometry. Two viable interfaces:
302
+
303
+ - **VAE/VPoser-like:** `z_t` is a latent pose code, `theta_t = D(z_t)`, with `E_latent = sum ||z_t||^2` and a temporal latent acceleration term. The decoder must output the local joint rotations used by FK.
304
+ - **Sequence prior:** a causal/bidirectional transformer or diffusion score supplies `-log p(theta_{0:T-1})`, evaluated on windows and summed with overlap.
305
+
306
+ The first is simpler and deterministic for an optimizer. Use the prior as a soft preference, not a hard decoder constraint, because the video may contain generated or unusual motion. Gate it down when the mask/keypoint residual is high; otherwise the prior can hallucinate a plausible action that is not the observed one.
307
+
308
+ ### 3.3 Joint limits
309
+
310
+ For each joint define limits in the local rest-relative frame. For a hinge, penalize angle outside `[lo, hi]`; for a ball joint, use a swing cone and twist interval. A differentiable soft barrier is
311
+
312
+ ```
313
+ B(z; lo,hi) = softplus((lo-z)/tau)^2 + softplus((z-hi)/tau)^2.
314
+ ```
315
+
316
+ For a swing-twist decomposition `R = R_swing R_twist`, use a cone barrier on swing angle and `B(twist; lo,hi)`. Apply a large final barrier or projection in the last optimization stage. Joint limits are a guardrail, not a substitute for data; over-tight limits are a likely source of failure for stylized motion.
317
+
318
+ ## 4. Scene terms
319
+
320
+ ### 4.1 Floor and stance detection
321
+
322
+ Represent the known floor as `n^T X + d = 0`, with `n` pointing upward. Define a sole contact point or a small set of sole vertices `F_{t,k}` for each foot. Do not use the ankle location as the contact point; use the skinned mesh's bottom sole geometry.
323
+
324
+ First compute a contact likelihood from the current trajectory:
325
+
326
+ ```
327
+ score_{t,f} = sigmoid(a0
328
+ - a1 * height(F_{t,f})
329
+ - a2 * ||F_{t,f}-F_{t-1,f}||
330
+ - a3 * image-foot-speed
331
+ + a4 * visible-foot-confidence).
332
+ ```
333
+
334
+ Then run a two-state HMM/Viterbi pass per foot with states `contact` and `swing`, transition penalties that discourage one-frame toggles, and a minimum stance duration. This is more stable than thresholding each frame independently. The contact state remains latent and can be updated every outer iteration.
335
+
336
+ For contact frames, maintain an anchor `A_{s,f}` per contiguous stance segment and use
337
+
338
+ ```
339
+ E_floor_contact = sum_{t,f} c_{t,f} ||F_{t,f}-A_{seg(t,f),f}||^2 / s_F^2
340
+ + lambda_height * c_{t,f} (n^T F_{t,f}+d)^2 / s_h^2.
341
+ ```
342
+
343
+ For every sole vertex, prevent floor penetration:
344
+
345
+ ```
346
+ E_floor_pen = sum_{t,v in sole} softplus(-(n^T V_t(v)+d-margin)/tau)^2.
347
+ ```
348
+
349
+ Use contact anchoring to correct the root and leg chain together. Root-only translation correction is exactly the mechanism that leaves foot skate in the current pipeline (`context.md`, lines 16-17).
350
+
351
+ ### 4.2 Box non-penetration and occlusion
352
+
353
+ For every body vertex `V_t(v)` and every solid box `b`, compute signed distance `sdf_b(V_t(v))`, positive outside the box. With a safety margin `m`, use
354
+
355
+ ```
356
+ E_box_pen = sum_{t,v,b} softplus((m - sdf_b(V_t(v)))/tau)^2.
357
+ ```
358
+
359
+ Sample all vertices for a low-resolution pass and collision-prone vertices/bone capsules for a high-resolution pass. If the solver supports constrained Gauss-Newton, impose `sdf_b(V_t(v)) >= m` as an inequality and solve with an augmented Lagrangian. Otherwise use the barrier above plus a final projection/repair pass. A box surface may be a valid hand or foot contact; non-penetration does not prohibit contact, it only prohibits the body entering the solid volume.
360
+
361
+ Render the boxes with depth before the character. For a character surface behind a box, use the visible render in `E_mask`, set corresponding keypoint visibility low, and rely on the motion prior, bone lengths, other visible joints, and contact constraints. Do not pull an occluded knee through the box merely to match an unreliable 2D detector.
362
+
363
+ An optional soft scene-contact term can attract a hand/foot to a box face only when image evidence supports it:
364
+
365
+ ```
366
+ E_box_contact = c_{t,hand} * robust(sdf_box(hand_point), 0)
367
+ ```
368
+
369
+ with a tangential velocity anchor during sustained contact. It must never reward penetration.
370
+
371
+ ### 4.3 Foot skate metric and penalty
372
+
373
+ For each inferred stance segment, penalize both translational and yaw/tangential sole motion. The position anchor above is the main term; additionally use
374
+
375
+ ```
376
+ E_skate = sum_{t,f} c_{t,f} ||P_t F_{t,f} - P_{t-1} F_{t-1,f}||^2,
377
+ ```
378
+
379
+ where `P_t` projects to the floor tangent plane. Use a robust penalty and turn it down at contact transitions. Keep a small image-data floor so a truly sliding foot is not frozen.
380
+
381
+ The complete prior/scene term is
382
+
383
+ ```
384
+ E_prior = E_acc_root + E_acc_rot + E_motion_prior + E_limits
385
+ + E_floor_contact + E_floor_pen + E_box_pen + E_skate.
386
+ ```
387
+
388
+ ## 5. Full objective and occlusion policy
389
+
390
+ For fixed assignment path `h` and contact states `c`, optimize
391
+
392
+ ```
393
+ E(q;h,c) = E_data(q;h)
394
+ + E_prior(q;c)
395
+ + E_init(q)
396
+ + E_boundary(q).
397
+ ```
398
+
399
+ `E_init` is a robust, decaying tether to GVHMR for the first optimization block only:
400
+
401
+ ```
402
+ E_init = alpha_init * sum_t ||x_t-x_t^gv||_rho^2
403
+ + alpha_init * sum_{t,j} d_SO3(R_{t,j},R^gv_{t,j})_rho^2.
404
+ ```
405
+
406
+ Decay `alpha_init` to zero; otherwise a bad GVHMR yaw/depth solution remains locked in. `E_boundary` is a weak prior on the first/last frame or known endpoint pose where the benchmark supplies one; do not use it on clips without endpoint evidence.
407
+
408
+ Occlusion policy is explicit:
409
+
410
+ 1. derive visibility by depth-buffering the known scene and current character;
411
+ 2. multiply keypoint and dense terms by visibility/confidence;
412
+ 3. compare silhouettes only in visible regions or with the scene-occluded render;
413
+ 4. increase temporal/prior/contact weight for hidden joints; and
414
+ 5. expose uncertainty/low-confidence flags for hidden segments.
415
+
416
+ A box-hidden leg is therefore inferred from the articulated state and constraints, not falsely observed from a keypoint behind the box.
417
+
418
+ ## 6. Solver and expected runtime
419
+
420
+ Use block-coordinate optimization over the whole clip:
421
+
422
+ 1. Run GVHMR and obtain `q^gv`, 2D keypoints, and initial confidence.
423
+ 2. Run SAM2 video segmentation and cache masks/confidences.
424
+ 3. Initialize depth/root with the known camera and bone scale; use the current GVHMR pose only as a starting point.
425
+ 4. Render visibility, estimate contacts, and run Viterbi for L/R assignments.
426
+ 5. Optimize all `q_0:T-1` jointly with Adam for 100-300 iterations, using coarse-to-fine masks and gradually increasing collision/contact weights.
427
+ 6. Run Viterbi assignment and contact HMM again.
428
+ 7. Refine with LBFGS (or damped Gauss-Newton/Levenberg-Marquardt) for 20-80 iterations with fixed assignments/contacts, then do one or two outer reassignment passes.
429
+ 8. Run a final collision/contact projection and calculate diagnostics.
430
+
431
+ Adam is forgiving when masks and assignments are changing. LBFGS or Gauss-Newton is better for the final coupled reprojection/contact fit. A pure per-frame optimizer is not acceptable: it cannot resolve a swap or an occluded leg using future and past frames.
432
+
433
+ Use temporal windows only for memory if necessary, with 10-20 frame overlap and state/velocity continuity constraints at boundaries. The preferred benchmark implementation is a full 120-240-frame batch because that is the requested behavior and because the clip is short enough for a 27-joint state.
434
+
435
+ Engineering estimate on a modern CUDA GPU, with 27 joints, a 10k-30k-vertex mesh, 512-ish mask resolution, and cached observations:
436
+
437
+ - 120 frames: roughly 30-120 seconds for keypoints + silhouette + collision, or 1-3 minutes if dense flow and high-resolution mesh rendering are enabled;
438
+ - 240 frames: roughly 1-4 minutes, or 3-8 minutes with dense flow/high-resolution collision.
439
+
440
+ These are planning estimates, not measured facts in the supplied context; benchmark them with a fixed seed and report wall-clock, GPU memory, and iteration count before adopting a production SLA. If differentiable full-mesh rendering is too slow, use a decimated mesh for optimization and the full mesh for the final collision/IoU pass.
441
+
442
+ ## 7. Automatic failure detection
443
+
444
+ Emit per-frame and per-clip diagnostics rather than a single score.
445
+
446
+ - **L/R ambiguity:** Viterbi best-vs-second-best margin small, frequent assignment switches, or two paths with similar data cost. Flag the affected bilateral groups and retain both candidate trajectories if the margin is below threshold.
447
+ - **Yaw flip / temporal discontinuity:** `angle(log(R_root(t-1)^T R_root(t)))` or pelvis image velocity exceeds calibrated limits; robust acceleration residual is an outlier. A high reprojection residual plus a large root jump indicates a bad local minimum rather than real motion.
448
+ - **Occlusion uncertainty:** low scene visibility, low SAM2 confidence, or a keypoint behind a box. Report low confidence and ensure the hidden-joint solution is prior/contact dominated.
449
+ - **Mask failure:** rendered-vs-observed area ratio outside a wide range, low boundary agreement, mask components not connected to the projected character, or sudden mask area/centroid jumps. Re-run SAM2 prompting on flagged frames or reduce mask weight; do not force the pose to fit a bad mask.
450
+ - **Box collision:** maximum and mean negative SDF, number of penetrating vertices, and penetration duration. Any persistent penetration above the benchmark tolerance is a hard failure even if IoU is good.
451
+ - **Floor/contact failure:** sole below floor, contact height residual, contact anchor residual, or foot tangent speed while in stance. Distinguish a bad contact label from a bad pose by re-running the contact HMM with collision fixed.
452
+ - **Foot skate:** stance-segment tangent speed and total drift in centimeters. Flag only sustained drift, not a contact transition.
453
+ - **Model mismatch:** structured residuals along the silhouette, high mask IoU failure despite low keypoint residual, or persistent joint-specific reprojection residual after assignment optimization. This means the known mesh/rig or segmentation is wrong, not that more smoothing is needed.
454
+ - **Over-smoothing:** acceleration energy far below the ground-truth distribution on the truth bench while data residual is not improving. The current run already shows limb acceleration below truth (`context.md`, lines 13-15), so this must be a tracked regression.
455
+ - **Global drift:** root trajectory changes substantially while image residual remains flat. Compare known floor/box relationships and the fixed-camera projection; on truth clips use root ATE as the definitive detector.
456
+
457
+ A clip is `accepted` only if collision, floor, assignment ambiguity, and mask/data residual thresholds pass. Otherwise export the best trajectory plus a machine-readable failure report and confidence intervals/flags.
458
+
459
+ ## 8. Measurement on the existing bench
460
+
461
+ Run the same 35 approved items and preserve the existing G0-G5/Gbest baselines (`context.md`, lines 9-11). Add the tracker as a new rung, not as a replacement that obscures regressions.
462
+
463
+ ### Ground-truth clips (8 gt renders plus known cube cases)
464
+
465
+ For each frame and the 17-joint comparison subset:
466
+
467
+ 1. **PA-MPJPE:** compute per-frame rigid Procrustes-aligned MPJPE in millimeters. This measures articulation independent of global translation/orientation.
468
+ 2. **Root ATE:** compare world root/pelvis positions with only the permitted initial alignment (not per-frame Procrustes); report RMSE, median, and 95th percentile in centimeters. Do not hide global drift with per-frame alignment.
469
+ 3. **Root rotation error:** geodesic SO(3) angle, median and 95th percentile.
470
+ 4. **2D reprojection error:** pixel RMSE/robust percentile on visible joints, with and without the latent assignment correction.
471
+ 5. **Silhouette IoU:** render the visible character with the known boxes and compare to the ground-truth character mask; report per-frame median, mean, and the fraction below the acceptance threshold.
472
+ 6. **Foot slide:** tangent-plane centimeters/second during ground-truth stance frames; use the truth contact labels for the primary metric and inferred contacts for the deployable metric.
473
+ 7. **Penetration:** maximum and mean body/box and sole/floor penetration in centimeters, plus violating-frame fraction.
474
+ 8. **Dynamics:** root/limb acceleration error and jerk, because the current baseline is over-smoothed.
475
+ 9. **Endpoint error:** pelvis and end-effector errors at the benchmark's A/B endpoints.
476
+ 10. **Runtime and memory:** wall-clock per 120/240 frames and peak GPU memory.
477
+
478
+ The primary go/no-go table should include PA-MPJPE, root ATE, IoU, foot slide, and penetration, exactly matching the requested bench. The latest reference values are roughly 64 mm PA-MPJPE, 29 cm root ATE, 0.585 gt IoU, and 3.6 cm/s foot slide for Gbest; the new method must beat these on a held-out aggregate without trading them for unacceptable penetration or over-smoothing (`context.md`, lines 13-15).
479
+
480
+ ### Fal-generated clips without truth
481
+
482
+ Use the existing proxy measures: 2D mask IoU, contact gap, penetration, and endpoint error A/B. Add assignment ambiguity, visibility-aware residuals, root jump count, and failure flags. The current fal reference is IoU 0.307, mean penetration 1.4 cm, pelvis jumps over 20 degrees/frame in 10 frames, and endpoint B 12 cm (`context.md`, lines 14-17). Do not call proxy improvement proof of 3D truth improvement; use the gt renders for that claim.
483
+
484
+ Use paired per-clip comparisons with fixed observations and report bootstrap confidence intervals across clips. Ablate at least:
485
+
486
+ - GVHMR only;
487
+ - + known camera/scale;
488
+ - + silhouette;
489
+ - + latent L/R Viterbi;
490
+ - + floor/contact;
491
+ - + box occlusion/collision;
492
+ - + learned prior;
493
+ - full model.
494
+
495
+ This identifies whether the claimed improvement comes from the structurally important terms rather than arbitrary smoothing. Also report the fraction of frames rejected by automatic diagnostics.
496
+
497
+ ## Honest comparison with a stronger regressor
498
+
499
+ This approach should beat a stronger regressor when the available facts are reliable and unusual: fixed calibrated camera, known body shape/lengths, known floor/boxes, long temporal context, and explicit contact/collision. It can resolve metric depth/scale, maintain one side assignment over time, keep a foot planted, and place an occluded limb on the physically valid side of a box. Those are precisely the information sources a generic framewise regressor does not enforce.
500
+
501
+ It will not automatically beat a stronger regressor on every frame. A modern video regressor may have a better learned image prior, better segmentation/features, and better hallucination of a fully hidden pose. Optimization can settle in a wrong local minimum if the mask is bad, the known camera/mesh is wrong, or the motion lies outside its prior. Dense flow is also weak on a uniform gray mannequin. The model-based fit cannot recover information that is absent and unconstrained; it only prevents physically impossible explanations.
502
+
503
+ Recommendation: do not replace GVHMR with a new regressor first. Keep it as initialization, implement the model-based batch refinement and diagnostics, and compare against any stronger video regressor under the same known-camera/scene benchmark. If the full fit still fails on visible gt clips, inspect residuals before buying a larger model: persistent structured residuals indicate mesh/camera/segmenter mismatch, while low-data/high-prior failures indicate that the learned motion prior or initializer needs improvement. Retraining a body-pose network remains an explicit later option, not a prerequisite for this design (`context.md`, lines 19-21).
504
+
505
+ ## Open assumptions to settle by measurement
506
+
507
+ 1. Whether SAM2's masks are stable enough on this specific faceless gray render; settle with mask precision/recall on the 8 gt renders and box clips.
508
+ 2. Whether global or pairwise L/R states are needed; settle by Viterbi margin and assignment accuracy against gt, with pairwise states capped to avoid overfitting.
509
+ 3. Whether the chosen cskel27 motion prior covers the user's generated actions; settle with held-out truth residuals and an ablation with the prior disabled.
510
+ 4. Actual runtime on the deployment GPU; settle with the fixed 120/240-frame benchmark described above.
511
+ 5. Exact collision tolerance and contact thresholds; calibrate from the known-truth cube/floor geometry, not visual preference.
512
+
513
+ The decision-complete first implementation is therefore: PyTorch/CUDA full-clip optimization, GVHMR initialization, SAM2 video masks, differentiable scene-aware rendering, robust keypoints, Viterbi L/R assignments, HMM stance contacts, temporal acceleration, weak learned prior, joint limits, floor/box constraints, and the existing bench as the acceptance gate.
@@ -0,0 +1,116 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * backfill-provenance.mjs - give sweep obs that never recorded their video a
4
+ * provenance manifest (#500, plan todo 22c).
5
+ *
6
+ * node tools/track/backfill-provenance.mjs --approved <approved.json> [--approved <more.json>] --obs-root <dir> [--dry-run]
7
+ *
8
+ * For every approved item and each <obs-root>/<set>/<name>/{base,g5}/obs.npz:
9
+ * - a manifest that already records videoSha256 (provenance absent or
10
+ * "recorded", e.g. publish-obs skin manifests) is left byte-for-byte alone;
11
+ * - otherwise the video is the one the manifest names, or (no manifest) the
12
+ * path the obs sweep extracted from - item.video ?? <dir>/<variant>/video.mp4,
13
+ * masks.mjs itemMaskInputs - and it must not be newer than the obs
14
+ * (masks.mjs checkObsVideo's `obs-video-newer` rule), else the item is refused;
15
+ * - manifest.json gains { video, videoSha256, obsSha256, detector (from
16
+ * extract.log when the manifest has none), provenance: "backfilled-path",
17
+ * backfilledAt, backfilledBy }. obs.npz is never written; its sha is checked
18
+ * unchanged after the manifest is written;
19
+ * - an earlier backfill with the same video/obs sha is left alone ("same");
20
+ * one that disagrees is refused.
21
+ * Exits 1 when any obs is refused.
22
+ */
23
+ import { createHash } from "node:crypto";
24
+ import { existsSync, readFileSync, statSync, writeFileSync } from "node:fs";
25
+ import { join, resolve } from "node:path";
26
+ import { fileURLToPath } from "node:url";
27
+ import { checkObsVideo, itemMaskInputs } from "./masks.mjs";
28
+
29
+ export const USAGE = "usage: node tools/track/backfill-provenance.mjs --approved <approved.json> [--approved <more.json>] --obs-root <dir> [--dry-run]";
30
+ export const KINDS = ["base", "g5"];
31
+ const BY = "tools/track/backfill-provenance.mjs";
32
+ const fileSha = (path) => createHash("sha256").update(readFileSync(path)).digest("hex");
33
+
34
+ export function parseArgs(argv) {
35
+ const o = { approved: [], dryRun: false };
36
+ for (let i = 0; i < argv.length; i += 1) {
37
+ const flag = argv[i];
38
+ if (flag === "--dry-run") { o.dryRun = true; continue; }
39
+ if (flag === "--help" || flag === "-h") { o.help = true; continue; }
40
+ if (flag !== "--approved" && flag !== "--obs-root") throw new Error(`unknown option ${flag}`);
41
+ const value = argv[++i];
42
+ if (value === undefined || value.startsWith("--")) throw new Error(`${flag} needs a value`);
43
+ if (flag === "--approved") o.approved.push(resolve(value)); else o.obsRoot = resolve(value);
44
+ }
45
+ if (!o.help && (!o.approved.length || !o.obsRoot)) throw new Error("--approved and --obs-root are required");
46
+ return o;
47
+ }
48
+
49
+ /** "palette" | "yolo" | ... from the wrapper's "detector: <name> selected" log line, or null. */
50
+ export function detectorFromLog(path) {
51
+ if (!existsSync(path)) return null;
52
+ return /detector: ([A-Za-z0-9_-]+) selected/.exec(readFileSync(path, "latin1"))?.[1] ?? null;
53
+ }
54
+
55
+ /** What to do for one obs dir; never writes. */
56
+ export function plan(item, kind, obsRoot, now = new Date().toISOString()) {
57
+ const label = `${item.set}/${item.name}/${kind}`;
58
+ const dir = join(obsRoot, item.set, item.name, kind), obs = join(dir, "obs.npz"), manifestPath = join(dir, "manifest.json");
59
+ if (!existsSync(obs)) return null;
60
+ let manifest = null;
61
+ if (existsSync(manifestPath)) {
62
+ try { manifest = JSON.parse(readFileSync(manifestPath, "utf8")); } catch (error) { return { item: label, status: "refused", reason: `unreadable manifest.json: ${error.message}` }; }
63
+ if (!manifest || typeof manifest !== "object" || Array.isArray(manifest)) return { item: label, status: "refused", reason: "manifest.json is not an object" };
64
+ if (manifest.source) return { item: label, status: "refused", reason: `manifest is a copy of ${manifest.source}, not an obs root manifest` };
65
+ if (manifest.videoSha256 && (manifest.provenance ?? "recorded") === "recorded") return { item: label, status: "recorded", video: manifest.video ?? null, videoSha256: manifest.videoSha256 };
66
+ }
67
+ const video = resolve(manifest?.video ?? itemMaskInputs(item, { obsRoot }).video);
68
+ if (!existsSync(video)) return { item: label, status: "refused", reason: `video ${video} is missing` };
69
+ const videoSha256 = fileSha(video), obsSha256 = fileSha(obs);
70
+ try { checkObsVideo({ origin: null, videoSha256, obsSha256, videoPath: video, videoMtimeMs: statSync(video).mtimeMs, obsMtimeMs: statSync(obs).mtimeMs }); }
71
+ catch (error) { return { item: label, status: "refused", reason: error.message }; }
72
+ if (manifest?.provenance === "backfilled-path") {
73
+ if (manifest.videoSha256 === videoSha256 && manifest.obsSha256 === obsSha256 && manifest.video === video) return { item: label, status: "same", video, videoSha256 };
74
+ return { item: label, status: "refused", reason: `existing backfill (video ${manifest.videoSha256}, obs ${manifest.obsSha256}) disagrees with ${video} (${videoSha256}) / obs ${obsSha256}` };
75
+ }
76
+ if (manifest && manifest.provenance !== undefined) return { item: label, status: "refused", reason: `manifest provenance ${manifest.provenance} is neither recorded nor backfilled-path` };
77
+ const detector = manifest?.detector ?? detectorFromLog(join(dir, "extract.log"));
78
+ const next = {
79
+ ...(manifest ?? {}),
80
+ video, videoSha256, obsSha256,
81
+ detector, ...(manifest?.detector ? {} : { detectorSource: detector ? "extract.log" : "unknown" }),
82
+ provenance: "backfilled-path", backfilledAt: now, backfilledBy: BY,
83
+ backfillRule: "video = the path the obs sweep extracted from (item.video ?? <dir>/<variant>/video.mp4), hashed after the fact; not newer than obs.npz",
84
+ };
85
+ return { item: label, status: "backfill", manifestPath, obs, obsSha256, video, videoSha256, detector, bytes: `${JSON.stringify(next, null, "\t")}\n` };
86
+ }
87
+
88
+ export function backfill({ approved, obsRoot, dryRun = false, now }) {
89
+ const items = approved.flatMap((path) => JSON.parse(readFileSync(path, "utf8")).items ?? []);
90
+ const results = [];
91
+ for (const item of items) for (const kind of KINDS) {
92
+ const p = plan(item, kind, obsRoot, now);
93
+ if (!p) continue;
94
+ if (p.status === "backfill" && !dryRun) {
95
+ writeFileSync(p.manifestPath, p.bytes);
96
+ if (fileSha(p.obs) !== p.obsSha256) throw new Error(`${p.obs} changed while backfilling`);
97
+ p.status = "backfilled";
98
+ }
99
+ const { bytes, ...rest } = p;
100
+ results.push(rest);
101
+ }
102
+ return { obsRoot, dryRun, results, ok: results.length > 0 && results.every((r) => r.status !== "refused") };
103
+ }
104
+
105
+ export function main(argv = process.argv.slice(2)) {
106
+ let options;
107
+ try { options = parseArgs(argv); } catch (error) { console.error(`backfill-provenance: ${error.message}\n${USAGE}`); return 2; }
108
+ if (options.help) { console.log(USAGE); return 0; }
109
+ const report = backfill(options);
110
+ for (const r of report.results) console.log(`${r.item}: ${r.status}${r.reason ? ` - ${r.reason}` : ""}${r.videoSha256 ? ` video=${r.video} sha=${r.videoSha256.slice(0, 12)}` : ""}${r.detector ? ` detector=${r.detector}` : ""}`);
111
+ const counts = report.results.reduce((acc, r) => ({ ...acc, [r.status]: (acc[r.status] ?? 0) + 1 }), {});
112
+ console.log(`backfill-provenance: ${Object.entries(counts).map(([k, v]) => `${k} ${v}`).join(", ") || "no obs found"}${options.dryRun ? " (dry run)" : ""}`);
113
+ return report.ok ? 0 : 1;
114
+ }
115
+
116
+ if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) process.exitCode = main();
@@ -0,0 +1,8 @@
1
+ /**
2
+ * The tracker receives a wall-clock allowance shared by bench and production.
3
+ * Keep this formula in one place so callers cannot drift apart.
4
+ */
5
+ export function trackerBudgetMs(frames) {
6
+ if (!Number.isInteger(frames) || frames < 0) throw new Error("frames must be a nonnegative integer");
7
+ return Math.max(300000, 1500 * frames + 120000);
8
+ }