reze-engine 0.50.3 → 0.50.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/dist/engine.d.ts +247 -2
  2. package/dist/engine.d.ts.map +1 -1
  3. package/dist/engine.js +822 -64
  4. package/dist/model.d.ts +49 -1
  5. package/dist/model.d.ts.map +1 -1
  6. package/dist/model.js +160 -4
  7. package/dist/physics/autofit.d.ts +147 -0
  8. package/dist/physics/autofit.d.ts.map +1 -0
  9. package/dist/physics/autofit.js +501 -0
  10. package/dist/physics/physics.d.ts +35 -0
  11. package/dist/physics/physics.d.ts.map +1 -1
  12. package/dist/physics/physics.js +64 -0
  13. package/dist/physics/world.d.ts +4 -0
  14. package/dist/physics/world.d.ts.map +1 -1
  15. package/dist/physics/world.js +6 -0
  16. package/dist/shaders/cast-api.d.ts +1 -1
  17. package/dist/shaders/cast-api.d.ts.map +1 -1
  18. package/dist/shaders/cast-layout.d.ts +44 -1
  19. package/dist/shaders/cast-layout.d.ts.map +1 -1
  20. package/dist/shaders/cast-layout.js +44 -1
  21. package/dist/shaders/materials/common.d.ts.map +1 -1
  22. package/dist/shaders/materials/common.js +7 -1
  23. package/dist/shaders/materials/nodes.d.ts +1 -1
  24. package/dist/shaders/materials/nodes.d.ts.map +1 -1
  25. package/dist/shaders/materials/nodes.js +17 -9
  26. package/dist/shaders/passes/composite.d.ts +1 -1
  27. package/dist/shaders/passes/composite.d.ts.map +1 -1
  28. package/dist/shaders/passes/depth-prepass.d.ts +1 -1
  29. package/dist/shaders/passes/depth-prepass.d.ts.map +1 -1
  30. package/dist/shaders/passes/depth-prepass.js +52 -12
  31. package/dist/shaders/passes/field-blit.d.ts +26 -0
  32. package/dist/shaders/passes/field-blit.d.ts.map +1 -0
  33. package/dist/shaders/passes/field-blit.js +65 -0
  34. package/dist/shaders/passes/ground-noise.d.ts +7 -0
  35. package/dist/shaders/passes/ground-noise.d.ts.map +1 -0
  36. package/dist/shaders/passes/ground-noise.js +88 -0
  37. package/dist/shaders/passes/ground.d.ts +16 -0
  38. package/dist/shaders/passes/ground.d.ts.map +1 -1
  39. package/dist/shaders/passes/ground.js +131 -27
  40. package/dist/shaders/passes/outline.d.ts +1 -1
  41. package/dist/shaders/passes/outline.d.ts.map +1 -1
  42. package/dist/shaders/passes/outline.js +12 -3
  43. package/dist/shaders/passes/particles.d.ts.map +1 -1
  44. package/dist/shaders/passes/particles.js +6 -2
  45. package/dist/shaders/passes/scene-contract.d.ts +38 -6
  46. package/dist/shaders/passes/scene-contract.d.ts.map +1 -1
  47. package/dist/shaders/passes/scene-contract.js +53 -16
  48. package/dist/shaders/passes/sim.d.ts +34 -0
  49. package/dist/shaders/passes/sim.d.ts.map +1 -0
  50. package/dist/shaders/passes/sim.js +169 -0
  51. package/dist/shaders/passes/trails.d.ts.map +1 -1
  52. package/dist/shaders/passes/trails.js +36 -8
  53. package/dist/shaders/score-api.d.ts +10 -0
  54. package/dist/shaders/score-api.d.ts.map +1 -0
  55. package/dist/shaders/score-api.js +114 -0
  56. package/package.json +1 -1
  57. package/src/engine.ts +867 -56
  58. package/src/model.ts +163 -4
  59. package/src/physics/physics.ts +63 -0
  60. package/src/physics/world.ts +7 -0
  61. package/src/shaders/cast-layout.ts +44 -1
  62. package/src/shaders/materials/common.ts +7 -1
  63. package/src/shaders/materials/nodes.ts +17 -9
  64. package/src/shaders/passes/depth-prepass.ts +53 -12
  65. package/src/shaders/passes/ground.ts +133 -27
  66. package/src/shaders/passes/outline.ts +14 -3
  67. package/src/shaders/passes/particles.ts +6 -2
  68. package/src/shaders/passes/scene-contract.ts +55 -16
  69. package/src/shaders/passes/trails.ts +36 -8
  70. package/dist/physics-debug.d.ts +0 -30
  71. package/dist/physics-debug.d.ts.map +0 -1
  72. package/dist/physics-debug.js +0 -526
  73. package/dist/shaders/materials/body.d.ts +0 -2
  74. package/dist/shaders/materials/body.d.ts.map +0 -1
  75. package/dist/shaders/materials/body.js +0 -95
  76. package/dist/shaders/materials/cloth_rough.d.ts +0 -2
  77. package/dist/shaders/materials/cloth_rough.d.ts.map +0 -1
  78. package/dist/shaders/materials/cloth_rough.js +0 -69
  79. package/dist/shaders/materials/cloth_smooth.d.ts +0 -2
  80. package/dist/shaders/materials/cloth_smooth.d.ts.map +0 -1
  81. package/dist/shaders/materials/cloth_smooth.js +0 -61
  82. package/dist/shaders/materials/default.d.ts +0 -2
  83. package/dist/shaders/materials/default.d.ts.map +0 -1
  84. package/dist/shaders/materials/default.js +0 -43
  85. package/dist/shaders/materials/eye.d.ts +0 -2
  86. package/dist/shaders/materials/eye.d.ts.map +0 -1
  87. package/dist/shaders/materials/eye.js +0 -60
  88. package/dist/shaders/materials/face.d.ts +0 -2
  89. package/dist/shaders/materials/face.d.ts.map +0 -1
  90. package/dist/shaders/materials/face.js +0 -95
  91. package/dist/shaders/materials/hair.d.ts +0 -2
  92. package/dist/shaders/materials/hair.d.ts.map +0 -1
  93. package/dist/shaders/materials/hair.js +0 -90
  94. package/dist/shaders/materials/metal.d.ts +0 -2
  95. package/dist/shaders/materials/metal.d.ts.map +0 -1
  96. package/dist/shaders/materials/metal.js +0 -77
  97. package/dist/shaders/materials/mmd_classic.d.ts +0 -2
  98. package/dist/shaders/materials/mmd_classic.d.ts.map +0 -1
  99. package/dist/shaders/materials/mmd_classic.js +0 -66
  100. package/dist/shaders/materials/stockings.d.ts +0 -2
  101. package/dist/shaders/materials/stockings.d.ts.map +0 -1
  102. package/dist/shaders/materials/stockings.js +0 -122
  103. package/dist/shaders/passes/physics-debug.d.ts +0 -2
  104. package/dist/shaders/passes/physics-debug.d.ts.map +0 -1
  105. package/dist/shaders/passes/physics-debug.js +0 -69
package/src/engine.ts CHANGED
@@ -50,9 +50,9 @@ import {
50
50
  hasLightEmit,
51
51
  parseLightCount,
52
52
  } from "./shaders/lights"
53
- import { groundShaderWgsl } from "./shaders/passes/ground"
54
- import { OUTLINE_SHADER_WGSL } from "./shaders/passes/outline"
55
- import { TRANSPARENT_DEPTH_PREPASS_WGSL } from "./shaders/passes/depth-prepass"
53
+ import { groundShaderWgsl, GROUND_NOISE_BAKE_WGSL, GROUND_NOISE_SIZE } from "./shaders/passes/ground"
54
+ import { outlineShaderWgsl } from "./shaders/passes/outline"
55
+ import { transparentDepthPrepassWgsl } from "./shaders/passes/depth-prepass"
56
56
  import { SELECTION_MASK_SHADER_WGSL, SELECTION_EDGE_SHADER_WGSL } from "./shaders/passes/selection"
57
57
  import { GIZMO_SHADER_WGSL } from "./shaders/passes/gizmo"
58
58
  import {
@@ -779,6 +779,28 @@ interface GpuMorph {
779
779
  // vertex sampling would misclassify hair (which must stay opaque-bucket for
780
780
  // stencil interplay and shadows).
781
781
 
782
+ /**
783
+ * A 2D context for the alpha readback, from whichever canvas this browser has.
784
+ *
785
+ * OffscreenCanvas's 2D context is not universal — Safari only gained it in
786
+ * 16.4, and a worker-less fallback has to be a DOM canvas. This used to be an
787
+ * unguarded `new OffscreenCanvas`, so a browser without it took the catch below
788
+ * and every material on the model was classified opaque. That is a rendering
789
+ * difference produced by a feature probe failing, which is the kind of thing
790
+ * that must never be silent.
791
+ */
792
+ function alphaReadbackContext(w: number, h: number): CanvasRenderingContext2D | OffscreenCanvasRenderingContext2D | null {
793
+ if (typeof OffscreenCanvas !== "undefined") {
794
+ const cx = new OffscreenCanvas(w, h).getContext("2d", { willReadFrequently: true })
795
+ if (cx) return cx
796
+ }
797
+ if (typeof document === "undefined") return null
798
+ const el = document.createElement("canvas")
799
+ el.width = w
800
+ el.height = h
801
+ return el.getContext("2d", { willReadFrequently: true })
802
+ }
803
+
782
804
  /** Downsampled alpha plane of a decoded texture (≤128², nearest-sampled). */
783
805
  function buildAlphaSampler(
784
806
  source: ImageBitmap | null,
@@ -789,20 +811,37 @@ function buildAlphaSampler(
789
811
  try {
790
812
  const w = Math.max(1, Math.min(128, width))
791
813
  const h = Math.max(1, Math.min(128, height))
792
- const canvas = new OffscreenCanvas(w, h)
793
- const cx = canvas.getContext("2d", { willReadFrequently: true })
794
- if (!cx) return null
795
- if (source) {
796
- cx.drawImage(source, 0, 0, w, h)
797
- } else if (rgba) {
798
- const tmp = new OffscreenCanvas(width, height)
799
- const tcx = tmp.getContext("2d")
800
- if (!tcx) return null
801
- tcx.putImageData(new ImageData(new Uint8ClampedArray(rgba), width, height), 0, 0)
802
- cx.drawImage(tmp, 0, 0, w, h)
803
- } else {
804
- return null
814
+ // Raw RGBA needs no canvas at all, and must not use one. It arrives from the
815
+ // TGA/DDS/PSD decoders as exact, straight-alpha bytes; the old path pushed it
816
+ // through putImageData → drawImage → getImageData, which is two premultiply
817
+ // round-trips and a resample to learn what was already in hand. Box-filtered
818
+ // straight off the array instead: same ≤128² plane, exact values, no canvas
819
+ // to be unavailable and no alpha to lose.
820
+ if (rgba) {
821
+ const a = new Uint8ClampedArray(w * h)
822
+ for (let y = 0; y < h; y++) {
823
+ const y0 = Math.floor((y * height) / h)
824
+ const y1 = Math.max(y0 + 1, Math.floor(((y + 1) * height) / h))
825
+ for (let x = 0; x < w; x++) {
826
+ const x0 = Math.floor((x * width) / w)
827
+ const x1 = Math.max(x0 + 1, Math.floor(((x + 1) * width) / w))
828
+ let sum = 0
829
+ let n = 0
830
+ for (let sy = y0; sy < y1; sy++) {
831
+ for (let sx = x0; sx < x1; sx++) {
832
+ sum += rgba[(sy * width + sx) * 4 + 3]
833
+ n++
834
+ }
835
+ }
836
+ a[y * w + x] = n > 0 ? sum / n : 255
837
+ }
838
+ }
839
+ return { a, w, h }
805
840
  }
841
+ if (!source) return null
842
+ const cx = alphaReadbackContext(w, h)
843
+ if (!cx) return null
844
+ cx.drawImage(source, 0, 0, w, h)
806
845
  const img = cx.getImageData(0, 0, w, h).data
807
846
  const a = new Uint8ClampedArray(w * h)
808
847
  for (let i = 0; i < w * h; i++) a[i] = img[i * 4 + 3]
@@ -1152,7 +1191,10 @@ interface EffectGrid {
1152
1191
  * ribbon read through rzTrail, so a trail costs one draw and nothing recorded.
1153
1192
  */
1154
1193
  interface EffectTrails {
1155
- instances: number
1194
+ /** Ribbons this effect declared — one per trailed anchor. The instance count
1195
+ * is derived from it per draw, against the live subject count, rather than
1196
+ * baked here against the four-subject cap. See drawTrails. */
1197
+ slots: number
1156
1198
  uniform: GPUBuffer
1157
1199
  data: Float32Array
1158
1200
  pipeline: GPURenderPipeline
@@ -1216,6 +1258,19 @@ interface EffectInstance {
1216
1258
  /** Mounted over the finished frame — and the reason the scene pass has to
1217
1259
  * STORE its depth, which it otherwise discards into tile memory. */
1218
1260
  hasForeground: boolean
1261
+ /** Does this source actually call rzObjectAt / rzMaterialAt?
1262
+ *
1263
+ * The exact sibling of hasForeground above, for the exact same reason. The id
1264
+ * attachment is the pass's most expensive STORE — rg16uint at the pass's
1265
+ * sample count, around 33MB a frame at 1080p — and it is written out for
1266
+ * every scene whether or not a single effect ever reads it. Declaring
1267
+ * the attachment is what keeps the pipelines agreeing; STORING it is what
1268
+ * costs, and only a reader can justify that.
1269
+ *
1270
+ * Parsed once at install rather than tested per frame: the answer cannot
1271
+ * change while an effect is installed, and the frame path should not be
1272
+ * running regexes. */
1273
+ readsIds: boolean
1219
1274
  /** Bones this source asked for, in ITS OWN declaration order. The scene table
1220
1275
  * maps these onto shared addresses; this list is what it is rebuilt from. */
1221
1276
  anchors: { bone: string; trail: boolean }[]
@@ -1287,7 +1342,9 @@ export class Engine {
1287
1342
  // Grouped materials use their group's own compiled pipeline.
1288
1343
  private neutralPipeline!: GPURenderPipeline
1289
1344
  private neutralPipelineNoDepthWrite!: GPURenderPipeline
1290
- private transparentDepthPrepassPipeline!: GPURenderPipeline
1345
+ private depthPrepassPipeline!: GPURenderPipeline
1346
+ private solidPrepassPipeline!: GPURenderPipeline
1347
+ private hairPrimePipeline!: GPURenderPipeline
1291
1348
  // ── Style group runtime ──
1292
1349
  // Shared 256 B zero StyleUniforms buffer (group(2) binding(4)) bound by every ungrouped
1293
1350
  // material; grouped materials rebind to their group's own buffer (per-model, in the
@@ -1372,6 +1429,23 @@ export class Engine {
1372
1429
  private multisampleTexture!: GPUTexture
1373
1430
  private hdrResolveTexture!: GPUTexture
1374
1431
  private static readonly MULTISAMPLE_COUNT = 4
1432
+ /**
1433
+ * Shadow map depth format — 16-bit, deliberately.
1434
+ *
1435
+ * The maps are ORTHOGRAPHIC, so depth is linear across the box: 65,536 steps
1436
+ * over the near cascade's 140-unit range is 0.002 units per step, and every
1437
+ * bias in play dwarfs it — the samplers subtract 0.0035 ndc (~229 of these
1438
+ * steps) and the materials offset along the normal by 0.08 units (~37 steps)
1439
+ * before the compare ever runs. Quantisation cannot flip an answer the biases
1440
+ * have already moved that far, so the pixels are identical to depth32float's.
1441
+ *
1442
+ * What is NOT identical is the bandwidth, which is the term WebKit pays
1443
+ * hardest: every PCF tap is a hardware-bilinear compare reading four texels,
1444
+ * so nine taps read half the bytes at 2 B/texel — 72 B/pixel instead of 144
1445
+ * across every shadowed surface on screen — and the 4096² map's clear+store
1446
+ * each frame drops from 64 MB to 32.
1447
+ */
1448
+ private static readonly SHADOW_DEPTH_FORMAT: GPUTextureFormat = "depth16unorm"
1375
1449
  // HDR intermediate format. rg11b10ufloat when the adapter exposes the
1376
1450
  // `rg11b10ufloat-renderable` feature (Chrome + Safari on Apple Silicon both
1377
1451
  // do), else fall back to rgba16float.
@@ -1451,6 +1525,16 @@ export class Engine {
1451
1525
  * this line, and the accessors then answer 0 rather than failing to compile.
1452
1526
  */
1453
1527
  private static readonly MRT_IDS = true
1528
+ /**
1529
+ * What fraction of its authored damping a chest rig's body keeps.
1530
+ *
1531
+ * The whole tuning surface for how long those rigs swing: lower rings
1532
+ * longer, 1 restores the authored value exactly. It does NOT change where
1533
+ * they hang at rest — that is the property that made damping the right knob
1534
+ * (see RezePhysics.setJiggleDamping). Judge it against the models that
1535
+ * motivated it; it is a starting point, not a measurement.
1536
+ */
1537
+ private static readonly JIGGLE_DAMPING_SCALE = 0.5
1454
1538
  /** The id attachment. Multisampled with the pass and NEVER resolved: an
1455
1539
  * averaged id belongs to nothing, so consumers textureLoad sample 0. */
1456
1540
  private idTexture: GPUTexture | null = null
@@ -1593,7 +1677,6 @@ export class Engine {
1593
1677
  private cullRebuilds = 0
1594
1678
  // ── Render bundles ──
1595
1679
  private opaqueBundle: GPURenderBundle | null = null
1596
- private transparentBundle: GPURenderBundle | null = null
1597
1680
  private shadowBundles: GPURenderBundle[] = []
1598
1681
  /** Set by scene STRUCTURE only. Every frame of animation, every physics step
1599
1682
  * and every camera move must leave this alone — re-recording constantly is
@@ -1617,7 +1700,34 @@ export class Engine {
1617
1700
  * field restructure moves. Restructuring it while it was the only untimed
1618
1701
  * pass in the frame would have meant reasoning about the cost instead of
1619
1702
  * reading it. */
1620
- private static readonly TIMED_PASSES = ["cull", "shadow", "scene", "field", "composite"] as const
1703
+ /**
1704
+ * The passes worth a number, in the order the frame runs them.
1705
+ *
1706
+ * These ARE the boxes on the architecture figure, deliberately: a reading that
1707
+ * cannot be pointed at a component is a reading nobody acts on. Three were
1708
+ * missing and each is a real per-frame cost a report of "it feels slower"
1709
+ * could have been about — the morph compute, the mirror's second pass over the
1710
+ * whole cast, and the bloom pyramid, which is NINE render passes and was the
1711
+ * largest unmeasured thing in the frame.
1712
+ *
1713
+ * The per-effect computes (particles, grids, lights) are deliberately absent:
1714
+ * they are a loop of one pass per effect, so there is no single span to stamp
1715
+ * and a number attributed to the wrong one is worse than no number. They fall
1716
+ * into the "rest" the readout derives from the frame time.
1717
+ *
1718
+ * Adding one costs two query slots and nothing else; the query set is sized
1719
+ * from this array's length.
1720
+ */
1721
+ private static readonly TIMED_PASSES = [
1722
+ "cull",
1723
+ "morph",
1724
+ "shadow",
1725
+ "mirror",
1726
+ "scene",
1727
+ "field",
1728
+ "bloom",
1729
+ "composite",
1730
+ ] as const
1621
1731
  private timestampQuerySet: GPUQuerySet | null = null
1622
1732
  private timestampResolve: GPUBuffer | null = null
1623
1733
  private timestampRead: GPUBuffer | null = null
@@ -1721,6 +1831,9 @@ export class Engine {
1721
1831
  * the plural is the whole point of this step and a singleton that has to be
1722
1832
  * "generalised later" is a singleton that shapes every call site against it.
1723
1833
  */
1834
+ /** Subjects the cast actually holds, set while it is filled. The ribbons size
1835
+ * their instance count by this rather than by the four-subject cap. */
1836
+ private castSubjectCount = 0
1724
1837
  private effects: EffectInstance[] = []
1725
1838
  /** The first installed effect, for the many places that legitimately want
1726
1839
  * "is anything installed" or the singleton API's one effect. */
@@ -2033,6 +2146,27 @@ export class Engine {
2033
2146
  }
2034
2147
  }
2035
2148
 
2149
+ /**
2150
+ * Whether bloom will actually reach the frame this frame.
2151
+ *
2152
+ * The composite multiplies the pyramid by this same effective intensity, so a
2153
+ * zero here means every pass that BUILDS the pyramid is work whose result is
2154
+ * multiplied by nothing. That was the state of it: `enabled` reached exactly
2155
+ * one line — the intensity uniform below — and the nine render passes that
2156
+ * fill the pyramid ran regardless, on every frame, of every scene, whether or
2157
+ * not anyone had asked for bloom.
2158
+ *
2159
+ * Nine passes is the number that matters rather than the pixels: on a
2160
+ * tile-based GPU a render pass is a tile load and store whatever it draws, so
2161
+ * this is paid in full on Apple hardware and largely hidden on a desktop
2162
+ * immediate-mode one. It is the same asymmetry as the bundle bug — cheap where
2163
+ * it was written, expensive where it was reported.
2164
+ */
2165
+ private bloomContributes(): boolean {
2166
+ const b = this.bloomSettings
2167
+ return b.enabled && b.intensity > 0
2168
+ }
2169
+
2036
2170
  private writeCompositeViewUniforms(): void {
2037
2171
  const v = this.viewTransform
2038
2172
  const b = this.bloomSettings
@@ -2343,6 +2477,9 @@ export class Engine {
2343
2477
  const lin = (c: number) => (c <= 0.04045 ? c / 12.92 : Math.pow((c + 0.055) / 1.055, 2.4))
2344
2478
  const atts = this.mirrorPassDescriptor.colorAttachments as GPURenderPassColorAttachment[]
2345
2479
  atts[0].clearValue = bg ? { r: lin(bg.x), g: lin(bg.y), b: lin(bg.z), a: 1 } : { r: 0, g: 0, b: 0, a: 0 }
2480
+ // The descriptor is reused every frame, so the stamp is set on it rather
2481
+ // than passed — same as the scene pass, which is built once too.
2482
+ this.mirrorPassDescriptor.timestampWrites = this.stamps("mirror")
2346
2483
  const pass = encoder.beginRenderPass(this.mirrorPassDescriptor)
2347
2484
  pass.setStencilReference(Engine.STENCIL_EYE_VALUE)
2348
2485
  const bundles: GPURenderBundle[] = []
@@ -2430,6 +2567,67 @@ export class Engine {
2430
2567
  return probe !== null && !err
2431
2568
  }
2432
2569
 
2570
+ /**
2571
+ * Record an uncaptured validation error, once per distinct message.
2572
+ *
2573
+ * Distinct, because the interesting property of these is WHICH ones happened,
2574
+ * not how many times — a pass that fails validation fails identically every
2575
+ * frame, so the second occurrence carries no information the first did not.
2576
+ * The count is kept anyway: "1×" and "94000×" distinguish a one-off at init
2577
+ * from something the render loop is doing, and that distinction is the first
2578
+ * question anyone reading the report will have.
2579
+ */
2580
+ private noteGpuError(message: string): void {
2581
+ const seen = this.gpuErrors.get(message)
2582
+ if (seen !== undefined) {
2583
+ this.gpuErrors.set(message, seen + 1)
2584
+ return
2585
+ }
2586
+ // The cap is on DISTINCT messages, so it is reached only by a device
2587
+ // disagreeing about many different things — at which point the first 32
2588
+ // have said what the device is, and the rest are noise.
2589
+ if (this.gpuErrors.size >= 32) return
2590
+ this.gpuErrors.set(message, 1)
2591
+ // First occurrence only, and console.error rather than a silent buffer: a
2592
+ // validation error means something did not draw, and a developer with the
2593
+ // console open should not have to know this report exists to find out.
2594
+ console.error(`[reze] WebGPU validation: ${message}`)
2595
+ }
2596
+
2597
+ /** Distinct uncaptured validation messages → how many times each arrived. */
2598
+ private readonly gpuErrors = new Map<string, number>()
2599
+
2600
+ /**
2601
+ * What this device actually gave us, and what it refused.
2602
+ *
2603
+ * The report exists because the three answers below are the ones that differ
2604
+ * between two browsers on the same machine, and a scene that renders wrong on
2605
+ * one of them is otherwise indistinguishable from a scene that is wrong. It is
2606
+ * meant to be read off a phone that cannot be attached to a debugger, which is
2607
+ * why it returns a value rather than logging: the host decides where to put it.
2608
+ */
2609
+ gpuReport(): {
2610
+ hdrFormat: GPUTextureFormat
2611
+ depthFormat: GPUTextureFormat
2612
+ reversedZ: boolean
2613
+ ids: boolean
2614
+ sampleCount: number
2615
+ presentationFormat: GPUTextureFormat
2616
+ features: string[]
2617
+ errors: { message: string; count: number }[]
2618
+ } {
2619
+ return {
2620
+ hdrFormat: this.hdrFormat,
2621
+ depthFormat: this.depthFormat,
2622
+ reversedZ: this.reversedZ,
2623
+ ids: mrtIdsEnabled(),
2624
+ sampleCount: Engine.MULTISAMPLE_COUNT,
2625
+ presentationFormat: this.presentationFormat,
2626
+ features: this.device ? [...this.device.features].sort() : [],
2627
+ errors: [...this.gpuErrors].map(([message, count]) => ({ message, count })),
2628
+ }
2629
+ }
2630
+
2433
2631
  private rebuildCompositeBindGroup(): void {
2434
2632
  if (!this.device || !this.hdrResolveTexture || !this.compositeBloomView || !this.depthReadView) return
2435
2633
  if (!this.castBuffer) return
@@ -3091,6 +3289,11 @@ export class Engine {
3091
3289
  paramsData,
3092
3290
  hasBackground,
3093
3291
  hasForeground,
3292
+ // The author's OWN source, not the assembled module: the assembled one
3293
+ // always carries the accessors (as real readers or as the zero stubs),
3294
+ // so matching against it would report every effect as a reader and the
3295
+ // attachment would be stored exactly as often as before.
3296
+ readsIds: /\brz(?:ObjectAt|MaterialAt)\s*\(/.test(wgsl),
3094
3297
  anchors,
3095
3298
  // The effect's own clock starts now. Per effect so that one installed
3096
3299
  // later still gets a frame where rzGridFrame() is 0 and can seed.
@@ -3690,7 +3893,10 @@ export class Engine {
3690
3893
  return {
3691
3894
  ok: true,
3692
3895
  state: {
3693
- instances: slots * MAX_EFFECT_SUBJECTS * (TRAIL_SAMPLES - 1) * TRAIL_SUBDIVISIONS,
3896
+ // Ribbons declared by this effect. The INSTANCE count is no longer
3897
+ // baked here — it follows the live subject count and is computed per
3898
+ // draw (see drawTrails).
3899
+ slots,
3694
3900
  uniform,
3695
3901
  data: new Float32Array(4),
3696
3902
  pipeline,
@@ -3752,13 +3958,27 @@ export class Engine {
3752
3958
  // The clock upload happens once, on the camera draw: queue writes land
3753
3959
  // before the encoder submits, so both passes read the same value — the
3754
3960
  // mirror draw writing it again would only write it twice.
3961
+ // Instances follow the LIVE subject count, not MAX_EFFECT_SUBJECTS.
3962
+ //
3963
+ // This used to be baked at install as slots x 4 x (samples-1) x subs, so a
3964
+ // scene with ONE character issued four characters' worth of ribbon quads
3965
+ // and threw three quarters of them away as degenerate — every frame, at
3966
+ // every sample length. Vertex invocations with no fragments are cheap, not
3967
+ // free, and they scale with the sample count, which is what made a longer
3968
+ // trail expensive.
3969
+ //
3970
+ // The shader decodes [ribbon][subject][segment] with the same number out
3971
+ // of its uniform, so the two cannot drift: change one without the other
3972
+ // and ribbons land on the wrong subject rather than merely costing more.
3973
+ const live = Math.max(1, this.castSubjectCount)
3755
3974
  if (view === "camera") {
3756
3975
  t.data[0] = this.sceneClock - e.epochScene
3976
+ t.data[1] = live
3757
3977
  this.device.queue.writeBuffer(t.uniform, 0, t.data.buffer as ArrayBuffer)
3758
3978
  }
3759
3979
  pass.setPipeline(t.pipeline)
3760
3980
  pass.setBindGroup(0, view === "mirror" ? t.mirrorBind : t.bind)
3761
- pass.draw(6, t.instances)
3981
+ pass.draw(6, t.slots * live * (TRAIL_SAMPLES - 1) * TRAIL_SUBDIVISIONS)
3762
3982
  }
3763
3983
  }
3764
3984
 
@@ -4187,6 +4407,26 @@ export class Engine {
4187
4407
  throw new Error("WebGPU is not supported in this browser.")
4188
4408
  }
4189
4409
  this.device = device
4410
+ // Every validation error this device ever raises, kept.
4411
+ //
4412
+ // WebGPU does not throw for a bad pipeline: createRenderPipeline hands back
4413
+ // an object that is already invalid, and the complaint arrives here instead
4414
+ // — or nowhere, if nobody is listening. Nobody was. That is why a device
4415
+ // that disagrees with this engine has, until now, had no way to say so: the
4416
+ // pipeline is built, setPipeline poisons the pass that uses it, and the
4417
+ // symptom reaches the user as geometry that is simply absent, with a clean
4418
+ // console. A browser is not obliged to agree with Dawn about what is legal,
4419
+ // and the two places this engine knowingly leans on Dawn's reading are both
4420
+ // in the scene pass (see scene-contract's writeMask-0 note).
4421
+ //
4422
+ // Bounded, and not on the console by default: a pass that fails validation
4423
+ // fails it again every frame, so an unbounded log is a memory leak with a
4424
+ // frame counter and an unconditional console.error is a browser tab that
4425
+ // stops responding. First N distinct messages, counted thereafter.
4426
+ device.addEventListener("uncapturederror", (e) => {
4427
+ const message = (e as GPUUncapturedErrorEvent).error.message
4428
+ this.noteGpuError(message)
4429
+ })
4190
4430
  if (hasRg11b10) this.hdrFormat = "rg11b10ufloat"
4191
4431
  // The override has the last word, including over a device that would have
4192
4432
  // been left on the fallback anyway — asking for the format you are already
@@ -4252,6 +4492,15 @@ export class Engine {
4252
4492
  this.createPipelines()
4253
4493
  this.setupResize()
4254
4494
  Engine.instance = this
4495
+ // One line, at init, naming the three answers that differ between two
4496
+ // browsers on the same machine. Not a debug flag and not a readout — it is
4497
+ // the identity of the renderer that was actually built, and on a device that
4498
+ // cannot be attached to a debugger it is the only way to know which of the
4499
+ // three paths is running. Every graphics application prints this.
4500
+ const r = this.gpuReport()
4501
+ console.info(
4502
+ `[reze] hdr=${r.hdrFormat} depth=${r.depthFormat} reversedZ=${r.reversedZ} ids=${r.ids} msaa=${r.sampleCount}`,
4503
+ )
4255
4504
  }
4256
4505
 
4257
4506
  // One-shot bake of EEVEE's combined BRDF LUT — DFG (bsdf_lut_frag.glsl) packed
@@ -4260,6 +4509,45 @@ export class Engine {
4260
4509
  // .ba = LTC magnitude → ltc_brdf_scale_from_lut
4261
4510
  // One texture fetch per fragment replaces the previous 2–3 taps. rgba8unorm
4262
4511
  // (vs rgba16float) halves sample bandwidth; DFG/LTC values fit [0,1] cleanly.
4512
+ /** The frost tile the ground samples instead of evaluating fbm per pixel. */
4513
+ private groundNoiseTexture!: GPUTexture
4514
+ private groundNoiseView!: GPUTextureView
4515
+
4516
+ /**
4517
+ * Bake the ground's frost noise once — the same fbm the shader used to run
4518
+ * per pixel, rendered to a seamless 1024² r8unorm tile at init.
4519
+ *
4520
+ * Why this exists is measured, not argued: on WebKit the ground's whole cost
4521
+ * was this evaluation (see the note at the sample site in ground.ts). The
4522
+ * bake is one fullscreen pass at init — under a millisecond, once — and the
4523
+ * per-pixel cost becomes a single level-0 texture read.
4524
+ */
4525
+ private bakeGroundNoise() {
4526
+ this.groundNoiseTexture = this.device.createTexture({
4527
+ label: "ground frost noise (baked)",
4528
+ size: [GROUND_NOISE_SIZE, GROUND_NOISE_SIZE],
4529
+ format: "r8unorm",
4530
+ usage: GPUTextureUsage.RENDER_ATTACHMENT | GPUTextureUsage.TEXTURE_BINDING,
4531
+ })
4532
+ this.groundNoiseView = this.groundNoiseTexture.createView()
4533
+ const module = this.device.createShaderModule({ label: "ground noise bake", code: GROUND_NOISE_BAKE_WGSL })
4534
+ const pipeline = this.device.createRenderPipeline({
4535
+ label: "ground noise bake",
4536
+ layout: "auto",
4537
+ vertex: { module, entryPoint: "vs" },
4538
+ fragment: { module, entryPoint: "fs", targets: [{ format: "r8unorm" }] },
4539
+ primitive: { topology: "triangle-list" },
4540
+ })
4541
+ const encoder = this.device.createCommandEncoder({ label: "ground noise bake" })
4542
+ const pass = encoder.beginRenderPass({
4543
+ colorAttachments: [{ view: this.groundNoiseView, loadOp: "clear", storeOp: "store" }],
4544
+ })
4545
+ pass.setPipeline(pipeline)
4546
+ pass.draw(3)
4547
+ pass.end()
4548
+ this.device.queue.submit([encoder.finish()])
4549
+ }
4550
+
4263
4551
  private bakeBrdfLut() {
4264
4552
  if (BRDF_LUT_SIZE !== LTC_MAG_LUT_SIZE) {
4265
4553
  throw new Error("BRDF LUT bake requires DFG size == LTC size (both 64).")
@@ -4798,23 +5086,62 @@ export class Engine {
4798
5086
  // occluded behind it. Color targets kept for pass compatibility, writeMask 0.
4799
5087
  const prepassModule = this.device.createShaderModule({
4800
5088
  label: "transparent depth prepass",
4801
- code: TRANSPARENT_DEPTH_PREPASS_WGSL,
5089
+ code: transparentDepthPrepassWgsl(),
4802
5090
  })
4803
- this.transparentDepthPrepassPipeline = this.device.createRenderPipeline({
4804
- label: "transparent depth prepass",
5091
+ const prepassDesc = {
4805
5092
  layout: mainPipelineLayout,
4806
5093
  vertex: { module: prepassModule, entryPoint: "vs", buffers: fullVertexBuffers as GPUVertexBufferLayout[] },
5094
+ primitive: { cullMode: "none" as GPUCullMode },
5095
+ multisample: { count: Engine.MULTISAMPLE_COUNT },
5096
+ depthStencil: {
5097
+ format: this.depthFormat,
5098
+ depthWriteEnabled: true,
5099
+ depthCompare: this.depthAhead,
5100
+ },
5101
+ }
5102
+ this.depthPrepassPipeline = this.device.createRenderPipeline({
5103
+ label: "opaque depth prepass",
5104
+ ...prepassDesc,
4807
5105
  fragment: {
4808
5106
  module: prepassModule,
4809
5107
  entryPoint: "fs",
4810
5108
  targets: sceneTargetsFor("depth-prepass", this.sceneFormats),
4811
5109
  },
4812
- primitive: { cullMode: "none" },
4813
- multisample: { count: Engine.MULTISAMPLE_COUNT },
5110
+ })
5111
+ // The SOLID prime: same module, cutoff forced to exactly 1.0. Only texels
5112
+ // whose blend ignores the destination may pre-claim depth in the
5113
+ // transparent phase — see the override's note in depth-prepass.ts.
5114
+ this.solidPrepassPipeline = this.device.createRenderPipeline({
5115
+ label: "transparent solid prepass",
5116
+ ...prepassDesc,
5117
+ fragment: {
5118
+ module: prepassModule,
5119
+ entryPoint: "fs",
5120
+ constants: { CUTOFF: 1.0 },
5121
+ targets: sceneTargetsFor("depth-prepass", this.sceneFormats),
5122
+ },
5123
+ })
5124
+ // The HAIR prime: solid texels only, and stencil-fenced off the eye
5125
+ // silhouette. It records after the non-hair opaque draws, so the eye has
5126
+ // already written its stencil — not-equal here is what keeps the primed
5127
+ // hair depth from ever claiming the pixels the see-through-hair pass needs
5128
+ // the eye to survive on. (Bundle draws use the PASS's stencil reference;
5129
+ // only pipeline/bind/vertex state resets across executeBundles.)
5130
+ this.hairPrimePipeline = this.device.createRenderPipeline({
5131
+ label: "hair depth prime",
5132
+ ...prepassDesc,
4814
5133
  depthStencil: {
4815
- format: this.depthFormat,
4816
- depthWriteEnabled: true,
4817
- depthCompare: this.depthAhead,
5134
+ ...prepassDesc.depthStencil,
5135
+ stencilFront: { compare: "not-equal", failOp: "keep", depthFailOp: "keep", passOp: "keep" },
5136
+ stencilBack: { compare: "not-equal", failOp: "keep", depthFailOp: "keep", passOp: "keep" },
5137
+ stencilReadMask: 0xff,
5138
+ stencilWriteMask: 0,
5139
+ },
5140
+ fragment: {
5141
+ module: prepassModule,
5142
+ entryPoint: "fs",
5143
+ constants: { CUTOFF: 1.0 },
5144
+ targets: sceneTargetsFor("depth-prepass", this.sceneFormats),
4818
5145
  },
4819
5146
  })
4820
5147
 
@@ -4852,7 +5179,7 @@ export class Engine {
4852
5179
  fragment: { module: shadowShader, entryPoint: "fs", targets: [] },
4853
5180
  primitive: { cullMode: "none" },
4854
5181
  depthStencil: {
4855
- format: "depth32float",
5182
+ format: Engine.SHADOW_DEPTH_FORMAT,
4856
5183
  depthWriteEnabled: true,
4857
5184
  depthCompare: "less-equal",
4858
5185
  // The shadow map keeps the NON-reversed convention (orthographicLh maps
@@ -4874,7 +5201,7 @@ export class Engine {
4874
5201
  this.device.createTexture({
4875
5202
  label: `shadow map cascade ${i}`,
4876
5203
  size: [c.mapSize, c.mapSize],
4877
- format: "depth32float",
5204
+ format: Engine.SHADOW_DEPTH_FORMAT,
4878
5205
  usage: GPUTextureUsage.RENDER_ATTACHMENT | GPUTextureUsage.TEXTURE_BINDING,
4879
5206
  }),
4880
5207
  )
@@ -4882,6 +5209,7 @@ export class Engine {
4882
5209
 
4883
5210
  // One-shot bake of Blender EEVEE's combined BRDF LUT (DFG + LTC packed rgba8unorm).
4884
5211
  this.bakeBrdfLut()
5212
+ this.bakeGroundNoise()
4885
5213
  this.agxFallbackTexture = this.device.createTexture({
4886
5214
  label: "AgX LUT fallback",
4887
5215
  size: [1, 1, 1],
@@ -4974,6 +5302,9 @@ export class Engine {
4974
5302
  { binding: 9, visibility: GPUShaderStage.FRAGMENT, texture: { sampleType: "float" } },
4975
5303
  { binding: 10, visibility: GPUShaderStage.FRAGMENT, sampler: {} },
4976
5304
  { binding: 11, visibility: GPUShaderStage.FRAGMENT, texture: { sampleType: "depth", multisampled: true } },
5305
+ // The baked frost tile — see bakeGroundNoise. Sampled with binding 10's
5306
+ // repeat sampler, so it brings no sampler of its own.
5307
+ { binding: 12, visibility: GPUShaderStage.FRAGMENT, texture: { sampleType: "float" } },
4977
5308
  ],
4978
5309
  })
4979
5310
  const groundShadowShader = this.device.createShaderModule({
@@ -5030,7 +5361,7 @@ export class Engine {
5030
5361
 
5031
5362
  const outlineShaderModule = this.device.createShaderModule({
5032
5363
  label: "outline shaders",
5033
- code: OUTLINE_SHADER_WGSL,
5364
+ code: outlineShaderWgsl(),
5034
5365
  })
5035
5366
 
5036
5367
  this.outlinePipeline = this.createRenderPipeline({
@@ -5518,6 +5849,21 @@ export class Engine {
5518
5849
  }
5519
5850
 
5520
5851
  private handleResize() {
5852
+ // No device, nothing to size.
5853
+ //
5854
+ // Three callers reach this, and two of them can arrive before init() has a
5855
+ // device or after teardown has released one: setRenderSize is PUBLIC and
5856
+ // unordered with respect to init, and the ResizeObserver keeps firing across
5857
+ // a hot reload while the replaced engine is still mounted. Both landed on
5858
+ // `this.device.createTexture` and threw — which is why this only shows up
5859
+ // during development, and why 0.43 never saw it: setRenderSize did not exist
5860
+ // to be called early.
5861
+ //
5862
+ // Returning is correct rather than merely quiet. fixedRenderSize has already
5863
+ // been recorded by the time we get here, and init() ends with its own
5864
+ // handleResize — so the size asked for before the device existed is applied
5865
+ // in full the moment there is something to apply it to.
5866
+ if (!this.device) return
5521
5867
  // Fixed override (offline/video rendering) wins; otherwise track CSS size × dpr.
5522
5868
  const dpr = window.devicePixelRatio || 1
5523
5869
  const width = this.fixedRenderSize ? this.fixedRenderSize.width : Math.floor(this.canvas.clientWidth * dpr)
@@ -6836,13 +7182,25 @@ export class Engine {
6836
7182
  return key
6837
7183
  }
6838
7184
 
7185
+ /** True while a stage is in the scene. Two things turn on it: the built-in
7186
+ * ground plane must not draw, and the far shadow cascade has nothing to
7187
+ * cover without one (see the cascade loop). */
7188
+ hasStage(): boolean {
7189
+ for (const inst of this.modelInstances.values()) if (inst.isStage) return true
7190
+ return false
7191
+ }
7192
+
6839
7193
  /** True while a stage is in the scene, which is when the built-in ground plane
6840
7194
  * must not draw. */
6841
7195
  groundIsSuppressed(): boolean {
6842
- for (const inst of this.modelInstances.values()) if (inst.isStage) return true
6843
- return false
7196
+ return this.hasStage()
6844
7197
  }
6845
7198
 
7199
+ /** Per cascade: does its map currently hold nothing but the cleared far plane?
7200
+ * Set by the cascade loop, which skips a cascade that is unwanted and already
7201
+ * cleared rather than re-clearing it every frame. */
7202
+ private shadowCascadeCleared: boolean[] = []
7203
+
6846
7204
  removeModel(name: string): void {
6847
7205
  const inst = this.modelInstances.get(name)
6848
7206
  if (!inst) return
@@ -7157,6 +7515,10 @@ export class Engine {
7157
7515
  if (inst.physics && this.physicsEnabled && inst.model.visible) {
7158
7516
  const tPhys = performance.now()
7159
7517
  inst.physics.step(deltaTime, inst.model.getWorldMatrices(), inst.model.getBoneInverseBindMatrices())
7518
+ // The step published new world matrices for the simulated bones; the
7519
+ // bones that INHERIT from them are still wearing the animated pose.
7520
+ // Returns immediately unless this rig actually has such a bone.
7521
+ inst.model.applyPhysicsAppend()
7160
7522
  physicsMs += performance.now() - tPhys
7161
7523
  }
7162
7524
  if (inst.vertexBufferNeedsUpdate) this.updateVertexBuffer(inst)
@@ -7205,7 +7567,7 @@ export class Engine {
7205
7567
  const gm = inst.gpuMorph
7206
7568
  if (!gm || !gm.dispatchNeeded) continue
7207
7569
  if (!pass) {
7208
- pass = encoder.beginComputePass({ label: "morph compute" })
7570
+ pass = encoder.beginComputePass({ label: "morph compute", timestampWrites: this.stamps("morph") })
7209
7571
  pass.setPipeline(this.morphComputePipeline)
7210
7572
  }
7211
7573
  pass.setBindGroup(0, gm.bindGroup)
@@ -7458,6 +7820,111 @@ export class Engine {
7458
7820
  flags[o + 20] = f
7459
7821
  }
7460
7822
  if (this.cullModelBuffer) this.device.queue.writeBuffer(this.cullModelBuffer, 0, data.buffer as ArrayBuffer)
7823
+ this.updateCasterSphere(data)
7824
+ }
7825
+
7826
+ /**
7827
+ * One sphere containing every shadow caster in the scene, for the ground.
7828
+ *
7829
+ * The ground's PCF is the most expensive thing in the frame on a tile-based
7830
+ * GPU — nine hardware-bilinear comparisons per pixel on a full-coverage draw,
7831
+ * which is what 0.33.2 was about and what a second cascade quietly undid. But
7832
+ * the floor is vastly larger than the thing standing on it, and a pixel the
7833
+ * character cannot possibly shadow does not need to ask the shadow map: the
7834
+ * answer is lit, and nine taps is an expensive way to spell it.
7835
+ *
7836
+ * So the ground gets a bound and tests against it in ALU. This reuses the
7837
+ * spheres the cull already builds every frame — an AABB over POSED bone
7838
+ * positions grown by the skin margin, which its own note calls a bound rather
7839
+ * than an estimate, so a jump or a physics-driven skirt is inside it by
7840
+ * construction. Union, not per model: one sphere is one test, and the ground
7841
+ * shader must not loop over the cast.
7842
+ *
7843
+ * A RIGID caster (a stage) leaves its cull sphere zeroed deliberately — the
7844
+ * cull reads its boxes instead — so any rigid model disables this entirely by
7845
+ * setting radius to -1. Wrong here is a missing shadow, and a scene with a
7846
+ * stage keeps the taps rather than risk one.
7847
+ */
7848
+ private updateCasterSphere(data: Float32Array): void {
7849
+ const out = this.casterSphere
7850
+ out[3] = 0
7851
+ let cx = 0
7852
+ let cy = 0
7853
+ let cz = 0
7854
+ let r = 0
7855
+ let any = false
7856
+ for (let i = 0; i < this.cullModels.length; i++) {
7857
+ const inst = this.cullModels[i]
7858
+ if (!inst.model.visible || inst.shadowDrawCalls.length === 0) continue
7859
+ if (inst.rigid) {
7860
+ // No sphere to read. Bail out of the whole optimisation.
7861
+ out[3] = -1
7862
+ return
7863
+ }
7864
+ const o = i * Engine.CULL_MODEL_FLOATS + 16
7865
+ const x = data[o]
7866
+ const y = data[o + 1]
7867
+ const z = data[o + 2]
7868
+ const rad = data[o + 3]
7869
+ if (rad <= 0) continue
7870
+ if (!any) {
7871
+ cx = x
7872
+ cy = y
7873
+ cz = z
7874
+ r = rad
7875
+ any = true
7876
+ continue
7877
+ }
7878
+ // Union of two spheres, the standard construction: if one already contains
7879
+ // the other keep it, else grow along the line between the centres.
7880
+ const dx = x - cx
7881
+ const dy = y - cy
7882
+ const dz = z - cz
7883
+ const d = Math.hypot(dx, dy, dz)
7884
+ if (d + rad <= r) continue
7885
+ if (d + r <= rad) {
7886
+ cx = x
7887
+ cy = y
7888
+ cz = z
7889
+ r = rad
7890
+ continue
7891
+ }
7892
+ const nr = (d + r + rad) * 0.5
7893
+ const t = (nr - r) / d
7894
+ cx += dx * t
7895
+ cy += dy * t
7896
+ cz += dz * t
7897
+ r = nr
7898
+ }
7899
+ out[0] = cx
7900
+ out[1] = cy
7901
+ out[2] = cz
7902
+ out[3] = any ? r : 0
7903
+ }
7904
+
7905
+ /** Every shadow caster in one sphere: (x, y, z, radius). radius 0 = nothing
7906
+ * casts, -1 = do not use (a rigid caster has no sphere). See updateCasterSphere. */
7907
+ private casterSphere = new Float32Array(4)
7908
+
7909
+ /** The ground's uniform block, kept so the caster sphere can be refreshed in
7910
+ * it every frame rather than rebuilding the buffer (addGround allocates). */
7911
+ private groundMaterialData: Float32Array | null = null
7912
+
7913
+ /**
7914
+ * Push this frame's caster sphere into the ground's uniform.
7915
+ *
7916
+ * Four floats, one writeBuffer, and only while a ground exists. Rebuilding the
7917
+ * block the way addGround does would allocate a buffer and a bind group per
7918
+ * frame, which is the cost this is trying to remove rather than a way to pay
7919
+ * it somewhere else.
7920
+ */
7921
+ private writeGroundCasterSphere(): void {
7922
+ const gb = this.groundMaterialData
7923
+ if (!gb || !this.groundShadowMaterialBuffer) return
7924
+ if (gb[20] === this.casterSphere[0] && gb[21] === this.casterSphere[1] &&
7925
+ gb[22] === this.casterSphere[2] && gb[23] === this.casterSphere[3]) return
7926
+ gb.set(this.casterSphere, 20)
7927
+ this.device.queue.writeBuffer(this.groundShadowMaterialBuffer, 80, this.casterSphere as Float32Array<ArrayBuffer>)
7461
7928
  }
7462
7929
 
7463
7930
  /**
@@ -7594,7 +8061,6 @@ export class Engine {
7594
8061
  }
7595
8062
  if (this.modelInstances.size === 0) {
7596
8063
  this.opaqueBundle = null
7597
- this.transparentBundle = null
7598
8064
  this.mirrorOpaqueBundle = null
7599
8065
  this.mirrorTransparentBundle = null
7600
8066
  this.shadowBundles = []
@@ -7606,14 +8072,15 @@ export class Engine {
7606
8072
  this.forEachInstance((inst) => this.renderModelOpaquePhase(opaque, inst, camView))
7607
8073
  this.opaqueBundle = opaque.finish({ label: "opaque phase" })
7608
8074
 
7609
- const transparent = this.device.createRenderBundleEncoder({ label: "transparent phase", ...scene })
7610
- this.forEachInstance((inst) => this.renderModelTransparentPhase(transparent, inst, camView))
7611
- this.transparentBundle = transparent.finish({ label: "transparent phase" })
7612
-
7613
- // The mirror pair: the same draws against the same formats, with the
7614
- // mirrored camera baked into bind group 0 and the mirror cull args baked
7615
- // into the indirect draws. Recorded whether or not a mirror is active —
7616
- // recording is cheap, and the bundles only execute when the pass runs.
8075
+ // NO camera transparent bundle. The camera pass draws that phase directly —
8076
+ // see the note at the executeBundles call for what recording one cost on
8077
+ // WebKit. Recording it anyway "in case" is not free and not harmless: it is
8078
+ // work on every rebuild, and a live bundle beside a direct draw of the same
8079
+ // phase is an invitation to execute it again.
8080
+ //
8081
+ // The MIRROR pair below keeps both bundles, and is allowed to: that pass
8082
+ // hands them to a single executeBundles with nothing direct in between,
8083
+ // which is the pattern that works.
7617
8084
  const mirrorView = this.sceneView("mirror")
7618
8085
  const mo = this.device.createRenderBundleEncoder({ label: "mirror opaque phase", ...scene })
7619
8086
  this.forEachInstance((inst) => this.renderModelOpaquePhase(mo, inst, mirrorView))
@@ -7631,7 +8098,7 @@ export class Engine {
7631
8098
  const shadow = this.device.createRenderBundleEncoder({
7632
8099
  label: `shadow pass, cascade ${ci}`,
7633
8100
  colorFormats: [],
7634
- depthStencilFormat: "depth32float",
8101
+ depthStencilFormat: Engine.SHADOW_DEPTH_FORMAT,
7635
8102
  })
7636
8103
  shadow.setPipeline(this.shadowDepthPipeline)
7637
8104
  this.forEachInstance((inst) => this.drawInstanceShadow(shadow, inst, ci))
@@ -7648,6 +8115,25 @@ export class Engine {
7648
8115
  return { querySet: this.timestampQuerySet, beginningOfPassWriteIndex: i * 2, endOfPassWriteIndex: i * 2 + 1 }
7649
8116
  }
7650
8117
 
8118
+ /**
8119
+ * Half a stamp, for a component that is several passes rather than one.
8120
+ *
8121
+ * Bloom is nine render passes — a prefilter blit, a downsample chain and an
8122
+ * upsample chain — and what anyone wants to know is what the PYRAMID cost, not
8123
+ * what its fourth mip cost. Both fields of GPURenderPassTimestampWrites are
8124
+ * optional, so the opening query goes on the first pass and the closing one on
8125
+ * the last, and the pair reads as one span across everything between.
8126
+ */
8127
+ private stampOpen(pass: (typeof Engine.TIMED_PASSES)[number]): GPURenderPassTimestampWrites | undefined {
8128
+ if (!this.timestampQuerySet) return undefined
8129
+ return { querySet: this.timestampQuerySet, beginningOfPassWriteIndex: Engine.TIMED_PASSES.indexOf(pass) * 2 }
8130
+ }
8131
+
8132
+ private stampClose(pass: (typeof Engine.TIMED_PASSES)[number]): GPURenderPassTimestampWrites | undefined {
8133
+ if (!this.timestampQuerySet) return undefined
8134
+ return { querySet: this.timestampQuerySet, endOfPassWriteIndex: Engine.TIMED_PASSES.indexOf(pass) * 2 + 1 }
8135
+ }
8136
+
7651
8137
  /**
7652
8138
  * Resolve this frame's timings and start a readback, at most one in flight.
7653
8139
  *
@@ -7659,6 +8145,8 @@ export class Engine {
7659
8145
  private resolveTimestamps(encoder: GPUCommandEncoder): void {
7660
8146
  const qs = this.timestampQuerySet
7661
8147
  if (!qs || !this.timestampResolve || !this.timestampRead) return
8148
+ // Nobody has asked. See getGpuTimings — the read is what enrols.
8149
+ if (!this.timestampsWanted) return
7662
8150
  const count = Engine.TIMED_PASSES.length * 2
7663
8151
  encoder.resolveQuerySet(qs, 0, count, this.timestampResolve, 0)
7664
8152
  if (this.timestampBusy) return
@@ -7698,11 +8186,27 @@ export class Engine {
7698
8186
  * The regression guard for the draw-path work: these are the numbers that say
7699
8187
  * whether restructuring cost anything, which is the claim being made — not
7700
8188
  * whether it made the scene faster, which was never the goal.
8189
+ *
8190
+ * ASKING IS WHAT TURNS IT ON. The first call to this enrols the engine in the
8191
+ * per-frame readback; until then resolveTimestamps does nothing. That is why
8192
+ * the first call returns null even on a device that can measure — the answer
8193
+ * arrives a frame or two later, which is already true of these numbers and
8194
+ * documented on resolveTimestamps.
8195
+ *
8196
+ * The alternative was what this used to do: resolve the query set, copy it to
8197
+ * a staging buffer and map that buffer, every frame, on every device, for a
8198
+ * reader that in this codebase did not exist. A map is a synchronisation point
8199
+ * and the whole path is instrumentation — paying for it unasked is the same
8200
+ * mistake as shipping a debug flag, only invisible.
7701
8201
  */
7702
8202
  getGpuTimings(): Record<string, number> | null {
8203
+ this.timestampsWanted = true
7703
8204
  return this.gpuPassMs
7704
8205
  }
7705
8206
 
8207
+ /** Set by the first getGpuTimings() call. See it for why asking is the switch. */
8208
+ private timestampsWanted = false
8209
+
7706
8210
  private dispatchCull(encoder: GPUCommandEncoder): void {
7707
8211
  if (this.cullListDirty) this.rebuildCullList()
7708
8212
  if (!this.cullBindGroup || this.cullDraws.length === 0) return
@@ -8032,6 +8536,20 @@ export class Engine {
8032
8536
  // solver for the heaviest mesh in the scene and dropping it afterwards was
8033
8537
  // both wasted work and an invariant maintained in the wrong place.
8034
8538
  const physics = !isStage && rbs.length > 0 ? new RezePhysics(rbs, model.getJoints()) : null
8539
+ // Which bones the simulation will overwrite, handed to the pose pipeline so
8540
+ // the append (付与) pass can consume the simulated result instead of the
8541
+ // animated one. Precomputed here, once, because the answer is topology —
8542
+ // see Model.setPhysicsDrivenBones for what it costs when a rig needs it and
8543
+ // why it costs nothing when none does.
8544
+ if (physics) {
8545
+ model.setPhysicsDrivenBones(physics.getPhysicsDrivenBones())
8546
+ // The bodies an inherited-from bone rides on are damped less than the
8547
+ // rest, so they swing longer WITHOUT hanging lower — see
8548
+ // RezePhysics.setJiggleDamping for why damping is the separable knob and
8549
+ // solver iterations are not.
8550
+ const appendSources = model.getAppendSourceBones()
8551
+ if (appendSources.length > 0) physics.setJiggleDamping(appendSources, Engine.JIGGLE_DAMPING_SCALE)
8552
+ }
8035
8553
  // Adopt the scene's air, or a model added mid-session would fall under
8036
8554
  // different gravity from the ones already on stage.
8037
8555
  if (physics) {
@@ -8297,7 +8815,8 @@ export class Engine {
8297
8815
  // Shadow map is already created in setupPipelines()
8298
8816
  // 20 floats: 16 for the original block, then (mirrorBlur, pad, pad, pad)
8299
8817
  // keeping the uniform vec4-aligned.
8300
- const gb = new Float32Array(20)
8818
+ const gb = new Float32Array(24)
8819
+ this.groundMaterialData = gb
8301
8820
  gb[0] = diffuseColor.x
8302
8821
  gb[1] = diffuseColor.y
8303
8822
  gb[2] = diffuseColor.z
@@ -8317,6 +8836,23 @@ export class Engine {
8317
8836
  this.groundMirror = gb[15]
8318
8837
  gb[16] = Math.min(Math.max(mirrorBlur, 0), 1)
8319
8838
  this.groundMirrorBlur = gb[16]
8839
+ // gb[17] — does the FAR cascade hold anything?
8840
+ //
8841
+ // It holds something only when a stage is loaded; that is what it exists for
8842
+ // and the cascade loop already skips drawing into it otherwise, leaving it
8843
+ // cleared. A cleared depth map compares as "no occluder", so the ground's far
8844
+ // branch is nine comparison taps whose answer is known in advance.
8845
+ //
8846
+ // That branch runs wherever the NEAR cascade does not reach, and the near one
8847
+ // is a 64-unit box around the camera target — so on a floor receding to the
8848
+ // horizon it is most of the visible pixels, on the most expensive
8849
+ // full-coverage draw in the frame. Skipping it is free in the exact sense:
8850
+ // the shader takes vis = 1.0, which is what the taps would have returned.
8851
+ gb[17] = this.hasStage() ? 1 : 0
8852
+ // gb[20..23] — the caster sphere, refreshed every frame by
8853
+ // writeGroundCasterSphere. Zero here so a frame that renders before the
8854
+ // first cull (there is one) reads "nothing casts" and skips the taps, which
8855
+ // is true: no model has been posed yet.
8320
8856
  this.groundShadowMaterialBuffer = this.device.createBuffer({
8321
8857
  size: gb.byteLength,
8322
8858
  usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST,
@@ -8351,6 +8887,7 @@ export class Engine {
8351
8887
  { binding: 9, resource: this.mirrorColorView! },
8352
8888
  { binding: 10, resource: this.materialSampler },
8353
8889
  { binding: 11, resource: this.mirrorDepthReadView! },
8890
+ { binding: 12, resource: this.groundNoiseView },
8354
8891
  ],
8355
8892
  })
8356
8893
  if (this.groundDrawCall) this.groundDrawCall.bindGroup = this.groundShadowBindGroup
@@ -8873,7 +9410,19 @@ export class Engine {
8873
9410
 
8874
9411
  // CPU alpha sampler for sheerness classification (see textureAlphaCache).
8875
9412
  // Canvas 2D premultiplies RGB on readback, but the ALPHA channel is exact.
8876
- this.textureAlphaCache.set(cacheKey, buildAlphaSampler(source, rgba, width, height))
9413
+ const alphaPlane = buildAlphaSampler(source, rgba, width, height)
9414
+ // Loud, because the fallback is WRONG rather than merely absent: a material
9415
+ // with no alpha plane scores avg 1 / translucentFrac 0, which routes sheer
9416
+ // fabric into the OPAQUE bucket and changes what the frame looks like. A
9417
+ // readback that fails is therefore a rendering bug, not a missing nicety,
9418
+ // and it must not reach the user as "the dress looks different on my phone".
9419
+ if (!alphaPlane) {
9420
+ console.warn(
9421
+ `[reze] alpha readback failed for ${cacheKey} — this material will be classified OPAQUE, ` +
9422
+ `so sheer fabric will not blend. The canvas 2D readback is what failed.`,
9423
+ )
9424
+ }
9425
+ this.textureAlphaCache.set(cacheKey, alphaPlane)
8877
9426
 
8878
9427
  const mipLevelCount = Math.floor(Math.log2(Math.max(width, height))) + 1
8879
9428
  const texture = this.device.createTexture({
@@ -9634,11 +10183,50 @@ export class Engine {
9634
10183
  const dofOn = this.depthOfField.enabled
9635
10184
  // ANY effect: one foreground mount anywhere in the scene, or one ribbon,
9636
10185
  // is enough to make the pass store its depth instead of discarding it.
9637
- const depthRead =
9638
- dofOn || this.effects.some((e) => e.hasForeground) || this.effects.some((e) => e.trails !== null)
10186
+ // Ribbons are NOT in this list, and removing them is the single largest
10187
+ // bandwidth saving in the frame on a tile-based GPU.
10188
+ //
10189
+ // They were, from when a ribbon was its own layer drawn after the scene and
10190
+ // depth-tested BY HAND against the stored buffer. That layer is gone —
10191
+ // ribbons draw inside this pass and the hardware depth test replaced what
10192
+ // they read it for (see trails.ts, "Binding 3 is GONE"). The clause outlived
10193
+ // the change by about twelve hours and then sat here.
10194
+ //
10195
+ // What it cost: this flag decides whether the pass STORES its depth or
10196
+ // discards it into tile memory, and the buffer is depth32float-stencil8 at
10197
+ // the pass's sample count — on a retina canvas that is a nine-figure number
10198
+ // of bytes written to RAM every frame, for a texture nothing then sampled.
10199
+ // Chrome hides it (an immediate-mode GPU has depth in memory regardless);
10200
+ // Apple's TBDR does not, which is exactly the reported shape: adding a hand
10201
+ // ribbon costs a lot of fps on Safari and almost nothing on Chrome.
10202
+ //
10203
+ // The two real readers are both in the composite and both have their own
10204
+ // flag above: linearDepth() feeds the DoF gather and the depth handed to a
10205
+ // foreground mount. Nothing else binds depthTex at all.
10206
+ const depthRead = dofOn || this.effects.some((e) => e.hasForeground)
9639
10207
  this.renderPassDescriptor.depthStencilAttachment!.depthStoreOp = depthRead ? "store" : "discard"
9640
10208
  if (depthRead) this.writeDepthOfFieldUniforms()
9641
10209
 
10210
+ // The id attachment, on exactly the same terms as the depth above it.
10211
+ //
10212
+ // It is the most expensive STORE in the pass — rg16uint at the pass's sample
10213
+ // count, ~33MB a frame at 1080p — and a uint target cannot be resolved, so
10214
+ // storing is the only way to get it out. It was stored unconditionally, for
10215
+ // every scene, whether or not anything read it. Nothing usually does: the
10216
+ // readers are rzObjectAt / rzMaterialAt in an effect that masks itself to one
10217
+ // character, and the id-buffer debug view.
10218
+ //
10219
+ // Discarding is not the same as removing. Every pipeline still declares the
10220
+ // attachment and the pass still carries it, so nothing is rebuilt and no
10221
+ // shader changes — the frame is bit-identical either way, because the only
10222
+ // difference is whether tile memory is written back to RAM after a pass
10223
+ // whose result no one is going to read.
10224
+ const idAtt = (this.renderPassDescriptor.colorAttachments as GPURenderPassColorAttachment[])[2]
10225
+ if (idAtt) {
10226
+ const idsRead = this.idDebug || this.effects.some((e) => e.readsIds)
10227
+ idAtt.storeOp = idsRead ? "store" : "discard"
10228
+ }
10229
+
9642
10230
  const encoder = this.device.createCommandEncoder()
9643
10231
 
9644
10232
  // GPU vertex morphs: write morphed positions into vertex buffers before any pass reads
@@ -9649,6 +10237,8 @@ export class Engine {
9649
10237
  // are settled, before the passes that draw from them.
9650
10238
  if (this.reflectionActive) this.updateMirrorCamera()
9651
10239
  if (hasModels) this.dispatchCull(encoder)
10240
+ // After the cull, which is what recomputes the spheres it unions.
10241
+ this.writeGroundCasterSphere()
9652
10242
 
9653
10243
  // After the cull, because a rebuild there can reallocate the argument
9654
10244
  // buffers and a bundle captures the buffer it recorded against.
@@ -9660,7 +10250,25 @@ export class Engine {
9660
10250
  // keeps PCF-sampling a character that is no longer in the scene. One clearing
9661
10251
  // pass on the transition to empty, then it stops.
9662
10252
  if (hasModels || this.shadowMapPopulated) {
10253
+ // The far cascade is the STAGE cascade, and it costs a full pass over the
10254
+ // whole cast every frame to say so. Its own spec explains what it is for —
10255
+ // "a set piece 100 units out still throws" — and a scene with no stage has
10256
+ // no set piece: every caster sits inside the near cascade's 64-unit box,
10257
+ // which follows the camera target, and the far map's only readers are
10258
+ // ground pixels beyond that box, where nothing is casting.
10259
+ //
10260
+ // So when no stage is loaded it is drawn ONCE, cleared, and then skipped —
10261
+ // the same shape as shadowMapPopulated above, and for the same reason. A
10262
+ // cleared depth map reads as "no occluder", which is the correct answer
10263
+ // here rather than a missing one. Load a stage and it comes straight back.
10264
+ //
10265
+ // 0.43 had ONE shadow map. This is half of what the second one costs.
10266
+ const stage = this.hasStage()
9663
10267
  for (let ci = 0; ci < SHADOW_CASCADES.length; ci++) {
10268
+ const wanted = ci === 0 || stage
10269
+ // Already cleared and still unwanted — nothing to do, and the map still
10270
+ // holds the far plane from the pass that cleared it.
10271
+ if (!wanted && this.shadowCascadeCleared[ci]) continue
9664
10272
  const sp = encoder.beginRenderPass({
9665
10273
  // One timestamp pair exists for "shadow"; the near cascade wears it.
9666
10274
  timestampWrites: ci === 0 ? this.stamps("shadow") : undefined,
@@ -9676,8 +10284,9 @@ export class Engine {
9676
10284
  // per-frame boolean, and baking it into a bundle would make toggling a
9677
10285
  // model re-record. It lives in the cull compute now, which zeroes the
9678
10286
  // instance count of an invisible model's draws.
9679
- if (this.shadowBundles[ci]) sp.executeBundles([this.shadowBundles[ci]])
10287
+ if (wanted && this.shadowBundles[ci]) sp.executeBundles([this.shadowBundles[ci]])
9680
10288
  sp.end()
10289
+ this.shadowCascadeCleared[ci] = !wanted
9681
10290
  }
9682
10291
  this.shadowMapPopulated = hasModels
9683
10292
  }
@@ -9708,8 +10317,44 @@ export class Engine {
9708
10317
  // anyway — eye writes it, hair tests not-equal, hairOverEyes tests equal.
9709
10318
  pass.setStencilReference(Engine.STENCIL_EYE_VALUE)
9710
10319
  if (this.opaqueBundle) pass.executeBundles([this.opaqueBundle])
10320
+ // Re-asserted after the bundle, not merely set once before it.
10321
+ //
10322
+ // Stencil reference is pass state a bundle cannot carry — GPURenderBundleEncoder
10323
+ // has no setStencilReference — which is why it was hoisted above the bundle in
10324
+ // the first place. But "cannot carry" and "cannot disturb" are different
10325
+ // claims, and only the first is specified. Everything below this line that
10326
+ // stencil-tests (hair at not-equal, outline hulls at not-equal) reads a
10327
+ // reference of 0 instead of 1 if a replay resets it, and not-equal against 0
10328
+ // is FALSE for the cleared buffer — every such fragment silently rejected.
10329
+ // One redundant word against a whole class of invisible failure.
10330
+ pass.setStencilReference(Engine.STENCIL_EYE_VALUE)
9711
10331
  if (this.hasGround) this.renderGround(pass)
9712
- if (this.transparentBundle) pass.executeBundles([this.transparentBundle])
10332
+ // The transparent phase is drawn DIRECTLY, and must stay that way. It is the
10333
+ // one part of this pass that is not bundled, so the reason is worth keeping.
10334
+ //
10335
+ // It WAS a bundle, and on WebKit the entire transparent bucket vanished while
10336
+ // the opaque bucket and the ground rendered perfectly — sheer fabric simply
10337
+ // absent, with no validation error anywhere. It was not the fragments: with
10338
+ // alpha forced to 1 they still never appeared, the cull reported every draw
10339
+ // visible with its GPU and CPU halves agreeing, and a cast model's
10340
+ // transparent draws use the SAME pipeline, bind groups and depth state as its
10341
+ // opaque ones (pipelineForDrawCall, forceDepthWrite). Identical draws,
10342
+ // identical state, one bucket rendering.
10343
+ //
10344
+ // What differed was only how they reached the pass: the opaque bundle is the
10345
+ // FIRST executeBundles here, and the transparent one was the SECOND, issued
10346
+ // after direct commands (the ground). Legal, and correct on Dawn. Not
10347
+ // replayed on WebKit. The mirror pass is the counter-example that pins the
10348
+ // shape of it — it passes BOTH bundles to a single executeBundles with
10349
+ // nothing direct in between, and has never lost a draw.
10350
+ //
10351
+ // So the rule this pass now keeps: at most one executeBundles, and nothing
10352
+ // direct before it. Bundling this phase again means first moving the ground
10353
+ // into the opaque bundle so the two can go in one call, the way the mirror
10354
+ // does it. The saving that buys is CPU encode time over a handful of draws,
10355
+ // which was never this renderer's bottleneck.
10356
+ const camView = this.sceneView("camera")
10357
+ this.forEachInstance((inst) => this.renderModelTransparentPhase(pass, inst, camView))
9713
10358
  // Last in the pass: depth-tested against everything drawn above, so a
9714
10359
  // particle behind the character is simply hidden, and still inside the HDR
9715
10360
  // target so an `@bloom` effect reaches the pyramid below.
@@ -9730,11 +10375,18 @@ export class Engine {
9730
10375
  // 3. Upsample (top-down): bloomUp[N-2] = tent(bloomDown[N-1]) + bloomDown[N-2],
9731
10376
  // then bloomUp[i] = tent(bloomUp[i+1]) + bloomDown[i] until i=0 (9-tap tent)
9732
10377
  // Composite reads bloomUp[0] and adds tint * intensity * bloom before Filmic.
9733
- if (this.bloomBlitBindGroup && this.compositeBindGroup && this.bloomMipCount > 0) {
10378
+ // bloomContributes() gates the whole pyramid, not just its intensity. The
10379
+ // composite still SAMPLES bloomUp[0] unconditionally, which is safe and
10380
+ // deliberate: it scales what it reads by the same effective intensity, so a
10381
+ // stale or never-written pyramid is multiplied by zero. Skipping the build
10382
+ // is therefore invisible in the frame and nine render passes cheaper.
10383
+ if (this.bloomContributes() && this.bloomBlitBindGroup && this.compositeBindGroup && this.bloomMipCount > 0) {
9734
10384
  const bloomAtt = this.bloomPassDescriptor.colorAttachments as GPURenderPassColorAttachment[]
9735
10385
 
9736
- // 1. Blit
10386
+ // 1. Blit — opens the pyramid's timing span. See stampOpen: the nine
10387
+ // passes below read as ONE component, which is the only useful grain.
9737
10388
  bloomAtt[0].view = this.bloomDownMipViews[0]
10389
+ this.bloomPassDescriptor.timestampWrites = this.stampOpen("bloom")
9738
10390
  const pBlit = encoder.beginRenderPass(this.bloomPassDescriptor)
9739
10391
  pBlit.setPipeline(this.bloomBlitPipeline)
9740
10392
  pBlit.setBindGroup(0, this.bloomBlitBindGroup)
@@ -9742,6 +10394,7 @@ export class Engine {
9742
10394
  pBlit.end()
9743
10395
 
9744
10396
  // 2. Downsample chain
10397
+ this.bloomPassDescriptor.timestampWrites = undefined
9745
10398
  for (let i = 1; i < this.bloomMipCount; i++) {
9746
10399
  bloomAtt[0].view = this.bloomDownMipViews[i]
9747
10400
  const p = encoder.beginRenderPass(this.bloomPassDescriptor)
@@ -9757,6 +10410,8 @@ export class Engine {
9757
10410
  for (let k = 0; k < upSteps; k++) {
9758
10411
  const levelIdx = topIdx - k // writes bloomUp[levelIdx]
9759
10412
  bloomAtt[0].view = this.bloomUpMipViews[levelIdx]
10413
+ // The LAST upsample closes the span opened on the blit.
10414
+ this.bloomPassDescriptor.timestampWrites = k === upSteps - 1 ? this.stampClose("bloom") : undefined
9760
10415
  const p = encoder.beginRenderPass(this.bloomPassDescriptor)
9761
10416
  p.setPipeline(this.bloomUpsamplePipeline)
9762
10417
  p.setBindGroup(0, this.bloomUpsampleBindGroups[k])
@@ -10325,16 +10980,29 @@ export class Engine {
10325
10980
  * makes outlines compose like MMD: every material drawn later in the author's
10326
10981
  * order covers earlier hulls, and each hull sits over everything drawn before it.
10327
10982
  */
10983
+ /** Is this draw's compiled class "hair"? Ungrouped draws never are — the
10984
+ * neutral pipeline is the auto class. */
10985
+ private isHairDraw(inst: ModelInstance, dc: DrawCall): boolean {
10986
+ if (!dc.groupId) return false
10987
+ const install = inst.styleGroups.get(dc.groupId)
10988
+ return install?.renderClass === "hair"
10989
+ }
10990
+
10328
10991
  private drawMaterials(
10329
10992
  pass: GPURenderPassEncoder | GPURenderBundleEncoder,
10330
10993
  inst: ModelInstance,
10331
10994
  type: "opaque" | "transparent",
10332
10995
  view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
10996
+ // The opaque phase walks its author order twice — non-hair, then hair — so
10997
+ // the hair depth prime can sit between the eye's stencil write and the hair
10998
+ // colour that must respect it. See renderModelOpaquePhase.
10999
+ only?: "hair" | "non-hair",
10333
11000
  ): void {
10334
11001
  let currentPipeline: GPURenderPipeline | null = null
10335
11002
  let bound = false
10336
11003
  for (const draw of inst.drawCalls) {
10337
11004
  if (draw.type !== type) continue
11005
+ if (only && (only === "hair") !== this.isHairDraw(inst, draw)) continue
10338
11006
  if (!bound) {
10339
11007
  pass.setBindGroup(0, view.perFrame)
10340
11008
  pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
@@ -10405,16 +11073,155 @@ export class Engine {
10405
11073
  view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
10406
11074
  ): void {
10407
11075
  this.setModelDrawState(pass, inst)
10408
- this.drawMaterials(pass, inst, "opaque", view)
11076
+ // Depth first, colour second — the close-up fix, and the oldest one there
11077
+ // is. See drawOpaqueDepthPrepass.
11078
+ this.drawOpaqueDepthPrepass(pass, inst, view)
11079
+ // The opaque author order, in two walks with the hair prime between them.
11080
+ //
11081
+ // Hair could not join the plain prepass: primed hair depth would depth-
11082
+ // reject the eye before it writes the stencil the see-through-hair pass
11083
+ // needs. But the trick only needs the eye BEFORE hair, not before
11084
+ // everything — so the non-hair walk runs first (the eye writes stencil
11085
+ // against real face depth, exactly as it always did), the prime then lays
11086
+ // hair depth down stencil-fenced off the eye silhouette, and the hair walk
11087
+ // shades once per pixel instead of once per card.
11088
+ //
11089
+ // The one thing this reorders: hair now draws after any opaque material
11090
+ // authored later than it. A soft hair edge over such a material blends
11091
+ // over the material instead of over whatever the framebuffer held mid-
11092
+ // order — deterministic where it used to be accidental, and only at
11093
+ // sub-alpha edge texels over late-authored geometry.
11094
+ this.drawMaterials(pass, inst, "opaque", view, "non-hair")
11095
+ this.drawHairDepthPrime(pass, inst, view)
11096
+ this.drawMaterials(pass, inst, "opaque", view, "hair")
10409
11097
  this.drawHairOverEyes(pass, inst, view)
10410
11098
  }
10411
11099
 
11100
+ /** Depth-only prime of the hair's alpha-1 texels, stencil-fenced off the eye
11101
+ * silhouette. See the note at its call site and hairPrimePipeline. */
11102
+ private drawHairDepthPrime(
11103
+ pass: GPURenderPassEncoder | GPURenderBundleEncoder,
11104
+ inst: ModelInstance,
11105
+ view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
11106
+ ): void {
11107
+ let bound = false
11108
+ for (const draw of inst.drawCalls) {
11109
+ if (draw.type !== "opaque" || !this.isHairDraw(inst, draw)) continue
11110
+ if (!bound) {
11111
+ pass.setPipeline(this.hairPrimePipeline)
11112
+ pass.setBindGroup(0, view.perFrame)
11113
+ pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
11114
+ bound = true
11115
+ }
11116
+ pass.setBindGroup(2, draw.bindGroup)
11117
+ this.issueDraw(pass, draw, view.args)
11118
+ }
11119
+ }
11120
+
11121
+ /**
11122
+ * Depth-only prime of the plain opaque draws, so each covered pixel SHADES
11123
+ * once instead of once per layer.
11124
+ *
11125
+ * The oldest fps complaint this engine has — zoom close and the frame drops,
11126
+ * in every material generation back to the earliest — was never the vertices
11127
+ * and never one shader's fault: with the fragment shaders flattened to a
11128
+ * constant the close-up ran smooth with identical geometry, overdraw and
11129
+ * MSAA. The cost is per-fragment shading TIMES how many times a pixel runs
11130
+ * it, and an MMD model at close-up is layers all the way down: cloth over
11131
+ * body, sleeves over cloth, hair over everything. Author-order drawing
11132
+ * shades every layer and then buries all but one.
11133
+ *
11134
+ * So the plain opaque draws lay their depth down first, through the same
11135
+ * depth-only pipeline the transparent bucket keeps for its dormant prepass —
11136
+ * same skinned vertex path (position marked @invariant in both modules, so
11137
+ * the colour pass lands on exactly these depths and its less-equal test
11138
+ * keeps the visible surface and rejects the buried ones), same alpha-0.5
11139
+ * cutout, writeMask 0 on every colour target. The pixels are identical by
11140
+ * construction: this pass writes no colour, and the colour pass draws
11141
+ * exactly what it always drew minus the fragments something opaque provably
11142
+ * covers.
11143
+ *
11144
+ * WHO IS IN. Only render-class "auto" with alpha-mode "opaque" — the body,
11145
+ * face and cloth materials that are the bulk of every model — plus every
11146
+ * ungrouped material (the neutral pipeline is that same class). WHO IS OUT,
11147
+ * each for a reason that would change pixels: EYE front-culls and gates on a
11148
+ * bone read, and pre-filled hair depth over the socket would depth-reject
11149
+ * the eye before it could write the stencil the see-through-hair pass needs
11150
+ * — which is also why HAIR stays out entirely. HASHED alpha (stockings)
11151
+ * discards by a position hash this pass does not run, so priming it would
11152
+ * punch its cutout into the depth buffer at the wrong texels. They all still
11153
+ * BENEFIT: their fragments early-z against the primed depth of whatever
11154
+ * plain opaque surface sits in front of them.
11155
+ */
11156
+ private drawOpaqueDepthPrepass(
11157
+ pass: GPURenderPassEncoder | GPURenderBundleEncoder,
11158
+ inst: ModelInstance,
11159
+ view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
11160
+ ): void {
11161
+ let bound = false
11162
+ for (const draw of inst.drawCalls) {
11163
+ if (draw.type !== "opaque") continue
11164
+ if (draw.groupId) {
11165
+ const install = inst.styleGroups.get(draw.groupId)
11166
+ if (install && !(install.renderClass === "auto" && install.alphaMode === "opaque")) continue
11167
+ }
11168
+ if (!bound) {
11169
+ pass.setPipeline(this.depthPrepassPipeline)
11170
+ pass.setBindGroup(0, view.perFrame)
11171
+ pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
11172
+ bound = true
11173
+ }
11174
+ pass.setBindGroup(2, draw.bindGroup)
11175
+ this.issueDraw(pass, draw, view.args)
11176
+ }
11177
+ }
11178
+
11179
+ /**
11180
+ * Depth-only prime of the transparent bucket's FULLY SOLID texels.
11181
+ *
11182
+ * The dress problem. A "transparent" MMD material is mostly weave at alpha
11183
+ * exactly 1 with sheer margins, and its layers draw in author order — so a
11184
+ * close-up skirt shades every buried panel and then covers the work. The
11185
+ * buried SHEER fragments must shade (their blend reads what is behind), but
11186
+ * at alpha 1 over-blending is plain replacement: the destination cannot
11187
+ * matter, so a fragment buried behind an alpha-1 texel contributes nothing.
11188
+ * Priming depth for exactly those texels (CUTOFF 1.0) rejects the buried
11189
+ * work and cannot move a pixel.
11190
+ *
11191
+ * A STAGE's transparent draws are excluded the way their colour path already
11192
+ * is: stage glass deliberately leaves depth alone so rain and particles
11193
+ * survive behind a dome (see pipelineForDrawCall), and a prime would put the
11194
+ * occlusion right back.
11195
+ */
11196
+ private drawTransparentSolidPrepass(
11197
+ pass: GPURenderPassEncoder | GPURenderBundleEncoder,
11198
+ inst: ModelInstance,
11199
+ view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
11200
+ ): void {
11201
+ if (inst.isStage) return
11202
+ let bound = false
11203
+ for (const draw of inst.drawCalls) {
11204
+ if (draw.type !== "transparent") continue
11205
+ if (!bound) {
11206
+ pass.setPipeline(this.solidPrepassPipeline)
11207
+ pass.setBindGroup(0, view.perFrame)
11208
+ pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
11209
+ bound = true
11210
+ }
11211
+ pass.setBindGroup(2, draw.bindGroup)
11212
+ this.issueDraw(pass, draw, view.args)
11213
+ }
11214
+ }
11215
+
10412
11216
  private renderModelTransparentPhase(
10413
11217
  pass: GPURenderPassEncoder | GPURenderBundleEncoder,
10414
11218
  inst: ModelInstance,
10415
11219
  view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
10416
11220
  ): void {
11221
+ // Draw state FIRST — each phase records into its own bundle encoder, and a
11222
+ // bundle starts with nothing bound.
10417
11223
  this.setModelDrawState(pass, inst)
11224
+ this.drawTransparentSolidPrepass(pass, inst, view)
10418
11225
  // Transparent: babylon-mmd's forceDepthWrite blending — PMX author order
10419
11226
  // with depth write ON. The accepted trade-off after trying every variant:
10420
11227
  // · depth-write ON (this): a fold hides its far side; rare view-dependent
@@ -10434,7 +11241,7 @@ export class Engine {
10434
11241
  for (const draw of inst.drawCalls) {
10435
11242
  if (draw.type !== "transparent") continue
10436
11243
  if (!bound) {
10437
- pass.setPipeline(this.transparentDepthPrepassPipeline)
11244
+ pass.setPipeline(this.depthPrepassPipeline)
10438
11245
  pass.setBindGroup(0, this.perFrameBindGroup)
10439
11246
  pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
10440
11247
  bound = true
@@ -10584,6 +11391,10 @@ export class Engine {
10584
11391
  n++
10585
11392
  })
10586
11393
  u[43] = n
11394
+ // The same number the ribbons size their instance count by — see
11395
+ // drawTrails. Recorded rather than recomputed: this loop is the one place
11396
+ // that knows how many subjects the cast actually ended up holding.
11397
+ this.castSubjectCount = n
10587
11398
  this.device.queue.writeBuffer(this.compositeUniformBuffer, 0, u)
10588
11399
  // Only what an effect declared, and only while one is installed. A scene
10589
11400
  // with no effect writes nothing here at all.