reze-engine 0.50.3 → 0.50.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/engine.d.ts +237 -2
- package/dist/engine.d.ts.map +1 -1
- package/dist/engine.js +793 -64
- package/dist/physics/autofit.d.ts +147 -0
- package/dist/physics/autofit.d.ts.map +1 -0
- package/dist/physics/autofit.js +501 -0
- package/dist/shaders/cast-api.d.ts +1 -1
- package/dist/shaders/cast-api.d.ts.map +1 -1
- package/dist/shaders/cast-layout.d.ts +44 -1
- package/dist/shaders/cast-layout.d.ts.map +1 -1
- package/dist/shaders/cast-layout.js +44 -1
- package/dist/shaders/materials/common.d.ts.map +1 -1
- package/dist/shaders/materials/common.js +7 -1
- package/dist/shaders/materials/nodes.d.ts +1 -1
- package/dist/shaders/materials/nodes.d.ts.map +1 -1
- package/dist/shaders/materials/nodes.js +17 -9
- package/dist/shaders/passes/composite.d.ts +1 -1
- package/dist/shaders/passes/composite.d.ts.map +1 -1
- package/dist/shaders/passes/depth-prepass.d.ts +1 -1
- package/dist/shaders/passes/depth-prepass.d.ts.map +1 -1
- package/dist/shaders/passes/depth-prepass.js +52 -12
- package/dist/shaders/passes/field-blit.d.ts +26 -0
- package/dist/shaders/passes/field-blit.d.ts.map +1 -0
- package/dist/shaders/passes/field-blit.js +65 -0
- package/dist/shaders/passes/ground-noise.d.ts +7 -0
- package/dist/shaders/passes/ground-noise.d.ts.map +1 -0
- package/dist/shaders/passes/ground-noise.js +88 -0
- package/dist/shaders/passes/ground.d.ts +16 -0
- package/dist/shaders/passes/ground.d.ts.map +1 -1
- package/dist/shaders/passes/ground.js +131 -27
- package/dist/shaders/passes/outline.d.ts +1 -1
- package/dist/shaders/passes/outline.d.ts.map +1 -1
- package/dist/shaders/passes/outline.js +12 -3
- package/dist/shaders/passes/particles.d.ts.map +1 -1
- package/dist/shaders/passes/particles.js +6 -2
- package/dist/shaders/passes/scene-contract.d.ts +38 -6
- package/dist/shaders/passes/scene-contract.d.ts.map +1 -1
- package/dist/shaders/passes/scene-contract.js +53 -16
- package/dist/shaders/passes/sim.d.ts +34 -0
- package/dist/shaders/passes/sim.d.ts.map +1 -0
- package/dist/shaders/passes/sim.js +169 -0
- package/dist/shaders/passes/trails.d.ts.map +1 -1
- package/dist/shaders/passes/trails.js +36 -8
- package/dist/shaders/score-api.d.ts +10 -0
- package/dist/shaders/score-api.d.ts.map +1 -0
- package/dist/shaders/score-api.js +114 -0
- package/package.json +1 -1
- package/src/engine.ts +839 -56
- package/src/shaders/cast-layout.ts +44 -1
- package/src/shaders/materials/common.ts +7 -1
- package/src/shaders/materials/nodes.ts +17 -9
- package/src/shaders/passes/depth-prepass.ts +53 -12
- package/src/shaders/passes/ground.ts +133 -27
- package/src/shaders/passes/outline.ts +14 -3
- package/src/shaders/passes/particles.ts +6 -2
- package/src/shaders/passes/scene-contract.ts +55 -16
- package/src/shaders/passes/trails.ts +36 -8
- package/dist/physics-debug.d.ts +0 -30
- package/dist/physics-debug.d.ts.map +0 -1
- package/dist/physics-debug.js +0 -526
- package/dist/shaders/materials/body.d.ts +0 -2
- package/dist/shaders/materials/body.d.ts.map +0 -1
- package/dist/shaders/materials/body.js +0 -95
- package/dist/shaders/materials/cloth_rough.d.ts +0 -2
- package/dist/shaders/materials/cloth_rough.d.ts.map +0 -1
- package/dist/shaders/materials/cloth_rough.js +0 -69
- package/dist/shaders/materials/cloth_smooth.d.ts +0 -2
- package/dist/shaders/materials/cloth_smooth.d.ts.map +0 -1
- package/dist/shaders/materials/cloth_smooth.js +0 -61
- package/dist/shaders/materials/default.d.ts +0 -2
- package/dist/shaders/materials/default.d.ts.map +0 -1
- package/dist/shaders/materials/default.js +0 -43
- package/dist/shaders/materials/eye.d.ts +0 -2
- package/dist/shaders/materials/eye.d.ts.map +0 -1
- package/dist/shaders/materials/eye.js +0 -60
- package/dist/shaders/materials/face.d.ts +0 -2
- package/dist/shaders/materials/face.d.ts.map +0 -1
- package/dist/shaders/materials/face.js +0 -95
- package/dist/shaders/materials/hair.d.ts +0 -2
- package/dist/shaders/materials/hair.d.ts.map +0 -1
- package/dist/shaders/materials/hair.js +0 -90
- package/dist/shaders/materials/metal.d.ts +0 -2
- package/dist/shaders/materials/metal.d.ts.map +0 -1
- package/dist/shaders/materials/metal.js +0 -77
- package/dist/shaders/materials/mmd_classic.d.ts +0 -2
- package/dist/shaders/materials/mmd_classic.d.ts.map +0 -1
- package/dist/shaders/materials/mmd_classic.js +0 -66
- package/dist/shaders/materials/stockings.d.ts +0 -2
- package/dist/shaders/materials/stockings.d.ts.map +0 -1
- package/dist/shaders/materials/stockings.js +0 -122
- package/dist/shaders/passes/physics-debug.d.ts +0 -2
- package/dist/shaders/passes/physics-debug.d.ts.map +0 -1
- package/dist/shaders/passes/physics-debug.js +0 -69
package/src/engine.ts
CHANGED
|
@@ -50,9 +50,9 @@ import {
|
|
|
50
50
|
hasLightEmit,
|
|
51
51
|
parseLightCount,
|
|
52
52
|
} from "./shaders/lights"
|
|
53
|
-
import { groundShaderWgsl } from "./shaders/passes/ground"
|
|
54
|
-
import {
|
|
55
|
-
import {
|
|
53
|
+
import { groundShaderWgsl, GROUND_NOISE_BAKE_WGSL, GROUND_NOISE_SIZE } from "./shaders/passes/ground"
|
|
54
|
+
import { outlineShaderWgsl } from "./shaders/passes/outline"
|
|
55
|
+
import { transparentDepthPrepassWgsl } from "./shaders/passes/depth-prepass"
|
|
56
56
|
import { SELECTION_MASK_SHADER_WGSL, SELECTION_EDGE_SHADER_WGSL } from "./shaders/passes/selection"
|
|
57
57
|
import { GIZMO_SHADER_WGSL } from "./shaders/passes/gizmo"
|
|
58
58
|
import {
|
|
@@ -779,6 +779,28 @@ interface GpuMorph {
|
|
|
779
779
|
// vertex sampling would misclassify hair (which must stay opaque-bucket for
|
|
780
780
|
// stencil interplay and shadows).
|
|
781
781
|
|
|
782
|
+
/**
|
|
783
|
+
* A 2D context for the alpha readback, from whichever canvas this browser has.
|
|
784
|
+
*
|
|
785
|
+
* OffscreenCanvas's 2D context is not universal — Safari only gained it in
|
|
786
|
+
* 16.4, and a worker-less fallback has to be a DOM canvas. This used to be an
|
|
787
|
+
* unguarded `new OffscreenCanvas`, so a browser without it took the catch below
|
|
788
|
+
* and every material on the model was classified opaque. That is a rendering
|
|
789
|
+
* difference produced by a feature probe failing, which is the kind of thing
|
|
790
|
+
* that must never be silent.
|
|
791
|
+
*/
|
|
792
|
+
function alphaReadbackContext(w: number, h: number): CanvasRenderingContext2D | OffscreenCanvasRenderingContext2D | null {
|
|
793
|
+
if (typeof OffscreenCanvas !== "undefined") {
|
|
794
|
+
const cx = new OffscreenCanvas(w, h).getContext("2d", { willReadFrequently: true })
|
|
795
|
+
if (cx) return cx
|
|
796
|
+
}
|
|
797
|
+
if (typeof document === "undefined") return null
|
|
798
|
+
const el = document.createElement("canvas")
|
|
799
|
+
el.width = w
|
|
800
|
+
el.height = h
|
|
801
|
+
return el.getContext("2d", { willReadFrequently: true })
|
|
802
|
+
}
|
|
803
|
+
|
|
782
804
|
/** Downsampled alpha plane of a decoded texture (≤128², nearest-sampled). */
|
|
783
805
|
function buildAlphaSampler(
|
|
784
806
|
source: ImageBitmap | null,
|
|
@@ -789,20 +811,37 @@ function buildAlphaSampler(
|
|
|
789
811
|
try {
|
|
790
812
|
const w = Math.max(1, Math.min(128, width))
|
|
791
813
|
const h = Math.max(1, Math.min(128, height))
|
|
792
|
-
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
const
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
814
|
+
// Raw RGBA needs no canvas at all, and must not use one. It arrives from the
|
|
815
|
+
// TGA/DDS/PSD decoders as exact, straight-alpha bytes; the old path pushed it
|
|
816
|
+
// through putImageData → drawImage → getImageData, which is two premultiply
|
|
817
|
+
// round-trips and a resample to learn what was already in hand. Box-filtered
|
|
818
|
+
// straight off the array instead: same ≤128² plane, exact values, no canvas
|
|
819
|
+
// to be unavailable and no alpha to lose.
|
|
820
|
+
if (rgba) {
|
|
821
|
+
const a = new Uint8ClampedArray(w * h)
|
|
822
|
+
for (let y = 0; y < h; y++) {
|
|
823
|
+
const y0 = Math.floor((y * height) / h)
|
|
824
|
+
const y1 = Math.max(y0 + 1, Math.floor(((y + 1) * height) / h))
|
|
825
|
+
for (let x = 0; x < w; x++) {
|
|
826
|
+
const x0 = Math.floor((x * width) / w)
|
|
827
|
+
const x1 = Math.max(x0 + 1, Math.floor(((x + 1) * width) / w))
|
|
828
|
+
let sum = 0
|
|
829
|
+
let n = 0
|
|
830
|
+
for (let sy = y0; sy < y1; sy++) {
|
|
831
|
+
for (let sx = x0; sx < x1; sx++) {
|
|
832
|
+
sum += rgba[(sy * width + sx) * 4 + 3]
|
|
833
|
+
n++
|
|
834
|
+
}
|
|
835
|
+
}
|
|
836
|
+
a[y * w + x] = n > 0 ? sum / n : 255
|
|
837
|
+
}
|
|
838
|
+
}
|
|
839
|
+
return { a, w, h }
|
|
805
840
|
}
|
|
841
|
+
if (!source) return null
|
|
842
|
+
const cx = alphaReadbackContext(w, h)
|
|
843
|
+
if (!cx) return null
|
|
844
|
+
cx.drawImage(source, 0, 0, w, h)
|
|
806
845
|
const img = cx.getImageData(0, 0, w, h).data
|
|
807
846
|
const a = new Uint8ClampedArray(w * h)
|
|
808
847
|
for (let i = 0; i < w * h; i++) a[i] = img[i * 4 + 3]
|
|
@@ -1152,7 +1191,10 @@ interface EffectGrid {
|
|
|
1152
1191
|
* ribbon read through rzTrail, so a trail costs one draw and nothing recorded.
|
|
1153
1192
|
*/
|
|
1154
1193
|
interface EffectTrails {
|
|
1155
|
-
|
|
1194
|
+
/** Ribbons this effect declared — one per trailed anchor. The instance count
|
|
1195
|
+
* is derived from it per draw, against the live subject count, rather than
|
|
1196
|
+
* baked here against the four-subject cap. See drawTrails. */
|
|
1197
|
+
slots: number
|
|
1156
1198
|
uniform: GPUBuffer
|
|
1157
1199
|
data: Float32Array
|
|
1158
1200
|
pipeline: GPURenderPipeline
|
|
@@ -1216,6 +1258,19 @@ interface EffectInstance {
|
|
|
1216
1258
|
/** Mounted over the finished frame — and the reason the scene pass has to
|
|
1217
1259
|
* STORE its depth, which it otherwise discards into tile memory. */
|
|
1218
1260
|
hasForeground: boolean
|
|
1261
|
+
/** Does this source actually call rzObjectAt / rzMaterialAt?
|
|
1262
|
+
*
|
|
1263
|
+
* The exact sibling of hasForeground above, for the exact same reason. The id
|
|
1264
|
+
* attachment is the pass's most expensive STORE — rg16uint at the pass's
|
|
1265
|
+
* sample count, around 33MB a frame at 1080p — and it is written out for
|
|
1266
|
+
* every scene whether or not a single effect ever reads it. Declaring
|
|
1267
|
+
* the attachment is what keeps the pipelines agreeing; STORING it is what
|
|
1268
|
+
* costs, and only a reader can justify that.
|
|
1269
|
+
*
|
|
1270
|
+
* Parsed once at install rather than tested per frame: the answer cannot
|
|
1271
|
+
* change while an effect is installed, and the frame path should not be
|
|
1272
|
+
* running regexes. */
|
|
1273
|
+
readsIds: boolean
|
|
1219
1274
|
/** Bones this source asked for, in ITS OWN declaration order. The scene table
|
|
1220
1275
|
* maps these onto shared addresses; this list is what it is rebuilt from. */
|
|
1221
1276
|
anchors: { bone: string; trail: boolean }[]
|
|
@@ -1287,7 +1342,9 @@ export class Engine {
|
|
|
1287
1342
|
// Grouped materials use their group's own compiled pipeline.
|
|
1288
1343
|
private neutralPipeline!: GPURenderPipeline
|
|
1289
1344
|
private neutralPipelineNoDepthWrite!: GPURenderPipeline
|
|
1290
|
-
private
|
|
1345
|
+
private depthPrepassPipeline!: GPURenderPipeline
|
|
1346
|
+
private solidPrepassPipeline!: GPURenderPipeline
|
|
1347
|
+
private hairPrimePipeline!: GPURenderPipeline
|
|
1291
1348
|
// ── Style group runtime ──
|
|
1292
1349
|
// Shared 256 B zero StyleUniforms buffer (group(2) binding(4)) bound by every ungrouped
|
|
1293
1350
|
// material; grouped materials rebind to their group's own buffer (per-model, in the
|
|
@@ -1372,6 +1429,23 @@ export class Engine {
|
|
|
1372
1429
|
private multisampleTexture!: GPUTexture
|
|
1373
1430
|
private hdrResolveTexture!: GPUTexture
|
|
1374
1431
|
private static readonly MULTISAMPLE_COUNT = 4
|
|
1432
|
+
/**
|
|
1433
|
+
* Shadow map depth format — 16-bit, deliberately.
|
|
1434
|
+
*
|
|
1435
|
+
* The maps are ORTHOGRAPHIC, so depth is linear across the box: 65,536 steps
|
|
1436
|
+
* over the near cascade's 140-unit range is 0.002 units per step, and every
|
|
1437
|
+
* bias in play dwarfs it — the samplers subtract 0.0035 ndc (~229 of these
|
|
1438
|
+
* steps) and the materials offset along the normal by 0.08 units (~37 steps)
|
|
1439
|
+
* before the compare ever runs. Quantisation cannot flip an answer the biases
|
|
1440
|
+
* have already moved that far, so the pixels are identical to depth32float's.
|
|
1441
|
+
*
|
|
1442
|
+
* What is NOT identical is the bandwidth, which is the term WebKit pays
|
|
1443
|
+
* hardest: every PCF tap is a hardware-bilinear compare reading four texels,
|
|
1444
|
+
* so nine taps read half the bytes at 2 B/texel — 72 B/pixel instead of 144
|
|
1445
|
+
* across every shadowed surface on screen — and the 4096² map's clear+store
|
|
1446
|
+
* each frame drops from 64 MB to 32.
|
|
1447
|
+
*/
|
|
1448
|
+
private static readonly SHADOW_DEPTH_FORMAT: GPUTextureFormat = "depth16unorm"
|
|
1375
1449
|
// HDR intermediate format. rg11b10ufloat when the adapter exposes the
|
|
1376
1450
|
// `rg11b10ufloat-renderable` feature (Chrome + Safari on Apple Silicon both
|
|
1377
1451
|
// do), else fall back to rgba16float.
|
|
@@ -1593,7 +1667,6 @@ export class Engine {
|
|
|
1593
1667
|
private cullRebuilds = 0
|
|
1594
1668
|
// ── Render bundles ──
|
|
1595
1669
|
private opaqueBundle: GPURenderBundle | null = null
|
|
1596
|
-
private transparentBundle: GPURenderBundle | null = null
|
|
1597
1670
|
private shadowBundles: GPURenderBundle[] = []
|
|
1598
1671
|
/** Set by scene STRUCTURE only. Every frame of animation, every physics step
|
|
1599
1672
|
* and every camera move must leave this alone — re-recording constantly is
|
|
@@ -1617,7 +1690,34 @@ export class Engine {
|
|
|
1617
1690
|
* field restructure moves. Restructuring it while it was the only untimed
|
|
1618
1691
|
* pass in the frame would have meant reasoning about the cost instead of
|
|
1619
1692
|
* reading it. */
|
|
1620
|
-
|
|
1693
|
+
/**
|
|
1694
|
+
* The passes worth a number, in the order the frame runs them.
|
|
1695
|
+
*
|
|
1696
|
+
* These ARE the boxes on the architecture figure, deliberately: a reading that
|
|
1697
|
+
* cannot be pointed at a component is a reading nobody acts on. Three were
|
|
1698
|
+
* missing and each is a real per-frame cost a report of "it feels slower"
|
|
1699
|
+
* could have been about — the morph compute, the mirror's second pass over the
|
|
1700
|
+
* whole cast, and the bloom pyramid, which is NINE render passes and was the
|
|
1701
|
+
* largest unmeasured thing in the frame.
|
|
1702
|
+
*
|
|
1703
|
+
* The per-effect computes (particles, grids, lights) are deliberately absent:
|
|
1704
|
+
* they are a loop of one pass per effect, so there is no single span to stamp
|
|
1705
|
+
* and a number attributed to the wrong one is worse than no number. They fall
|
|
1706
|
+
* into the "rest" the readout derives from the frame time.
|
|
1707
|
+
*
|
|
1708
|
+
* Adding one costs two query slots and nothing else; the query set is sized
|
|
1709
|
+
* from this array's length.
|
|
1710
|
+
*/
|
|
1711
|
+
private static readonly TIMED_PASSES = [
|
|
1712
|
+
"cull",
|
|
1713
|
+
"morph",
|
|
1714
|
+
"shadow",
|
|
1715
|
+
"mirror",
|
|
1716
|
+
"scene",
|
|
1717
|
+
"field",
|
|
1718
|
+
"bloom",
|
|
1719
|
+
"composite",
|
|
1720
|
+
] as const
|
|
1621
1721
|
private timestampQuerySet: GPUQuerySet | null = null
|
|
1622
1722
|
private timestampResolve: GPUBuffer | null = null
|
|
1623
1723
|
private timestampRead: GPUBuffer | null = null
|
|
@@ -1721,6 +1821,9 @@ export class Engine {
|
|
|
1721
1821
|
* the plural is the whole point of this step and a singleton that has to be
|
|
1722
1822
|
* "generalised later" is a singleton that shapes every call site against it.
|
|
1723
1823
|
*/
|
|
1824
|
+
/** Subjects the cast actually holds, set while it is filled. The ribbons size
|
|
1825
|
+
* their instance count by this rather than by the four-subject cap. */
|
|
1826
|
+
private castSubjectCount = 0
|
|
1724
1827
|
private effects: EffectInstance[] = []
|
|
1725
1828
|
/** The first installed effect, for the many places that legitimately want
|
|
1726
1829
|
* "is anything installed" or the singleton API's one effect. */
|
|
@@ -2033,6 +2136,27 @@ export class Engine {
|
|
|
2033
2136
|
}
|
|
2034
2137
|
}
|
|
2035
2138
|
|
|
2139
|
+
/**
|
|
2140
|
+
* Whether bloom will actually reach the frame this frame.
|
|
2141
|
+
*
|
|
2142
|
+
* The composite multiplies the pyramid by this same effective intensity, so a
|
|
2143
|
+
* zero here means every pass that BUILDS the pyramid is work whose result is
|
|
2144
|
+
* multiplied by nothing. That was the state of it: `enabled` reached exactly
|
|
2145
|
+
* one line — the intensity uniform below — and the nine render passes that
|
|
2146
|
+
* fill the pyramid ran regardless, on every frame, of every scene, whether or
|
|
2147
|
+
* not anyone had asked for bloom.
|
|
2148
|
+
*
|
|
2149
|
+
* Nine passes is the number that matters rather than the pixels: on a
|
|
2150
|
+
* tile-based GPU a render pass is a tile load and store whatever it draws, so
|
|
2151
|
+
* this is paid in full on Apple hardware and largely hidden on a desktop
|
|
2152
|
+
* immediate-mode one. It is the same asymmetry as the bundle bug — cheap where
|
|
2153
|
+
* it was written, expensive where it was reported.
|
|
2154
|
+
*/
|
|
2155
|
+
private bloomContributes(): boolean {
|
|
2156
|
+
const b = this.bloomSettings
|
|
2157
|
+
return b.enabled && b.intensity > 0
|
|
2158
|
+
}
|
|
2159
|
+
|
|
2036
2160
|
private writeCompositeViewUniforms(): void {
|
|
2037
2161
|
const v = this.viewTransform
|
|
2038
2162
|
const b = this.bloomSettings
|
|
@@ -2343,6 +2467,9 @@ export class Engine {
|
|
|
2343
2467
|
const lin = (c: number) => (c <= 0.04045 ? c / 12.92 : Math.pow((c + 0.055) / 1.055, 2.4))
|
|
2344
2468
|
const atts = this.mirrorPassDescriptor.colorAttachments as GPURenderPassColorAttachment[]
|
|
2345
2469
|
atts[0].clearValue = bg ? { r: lin(bg.x), g: lin(bg.y), b: lin(bg.z), a: 1 } : { r: 0, g: 0, b: 0, a: 0 }
|
|
2470
|
+
// The descriptor is reused every frame, so the stamp is set on it rather
|
|
2471
|
+
// than passed — same as the scene pass, which is built once too.
|
|
2472
|
+
this.mirrorPassDescriptor.timestampWrites = this.stamps("mirror")
|
|
2346
2473
|
const pass = encoder.beginRenderPass(this.mirrorPassDescriptor)
|
|
2347
2474
|
pass.setStencilReference(Engine.STENCIL_EYE_VALUE)
|
|
2348
2475
|
const bundles: GPURenderBundle[] = []
|
|
@@ -2430,6 +2557,67 @@ export class Engine {
|
|
|
2430
2557
|
return probe !== null && !err
|
|
2431
2558
|
}
|
|
2432
2559
|
|
|
2560
|
+
/**
|
|
2561
|
+
* Record an uncaptured validation error, once per distinct message.
|
|
2562
|
+
*
|
|
2563
|
+
* Distinct, because the interesting property of these is WHICH ones happened,
|
|
2564
|
+
* not how many times — a pass that fails validation fails identically every
|
|
2565
|
+
* frame, so the second occurrence carries no information the first did not.
|
|
2566
|
+
* The count is kept anyway: "1×" and "94000×" distinguish a one-off at init
|
|
2567
|
+
* from something the render loop is doing, and that distinction is the first
|
|
2568
|
+
* question anyone reading the report will have.
|
|
2569
|
+
*/
|
|
2570
|
+
private noteGpuError(message: string): void {
|
|
2571
|
+
const seen = this.gpuErrors.get(message)
|
|
2572
|
+
if (seen !== undefined) {
|
|
2573
|
+
this.gpuErrors.set(message, seen + 1)
|
|
2574
|
+
return
|
|
2575
|
+
}
|
|
2576
|
+
// The cap is on DISTINCT messages, so it is reached only by a device
|
|
2577
|
+
// disagreeing about many different things — at which point the first 32
|
|
2578
|
+
// have said what the device is, and the rest are noise.
|
|
2579
|
+
if (this.gpuErrors.size >= 32) return
|
|
2580
|
+
this.gpuErrors.set(message, 1)
|
|
2581
|
+
// First occurrence only, and console.error rather than a silent buffer: a
|
|
2582
|
+
// validation error means something did not draw, and a developer with the
|
|
2583
|
+
// console open should not have to know this report exists to find out.
|
|
2584
|
+
console.error(`[reze] WebGPU validation: ${message}`)
|
|
2585
|
+
}
|
|
2586
|
+
|
|
2587
|
+
/** Distinct uncaptured validation messages → how many times each arrived. */
|
|
2588
|
+
private readonly gpuErrors = new Map<string, number>()
|
|
2589
|
+
|
|
2590
|
+
/**
|
|
2591
|
+
* What this device actually gave us, and what it refused.
|
|
2592
|
+
*
|
|
2593
|
+
* The report exists because the three answers below are the ones that differ
|
|
2594
|
+
* between two browsers on the same machine, and a scene that renders wrong on
|
|
2595
|
+
* one of them is otherwise indistinguishable from a scene that is wrong. It is
|
|
2596
|
+
* meant to be read off a phone that cannot be attached to a debugger, which is
|
|
2597
|
+
* why it returns a value rather than logging: the host decides where to put it.
|
|
2598
|
+
*/
|
|
2599
|
+
gpuReport(): {
|
|
2600
|
+
hdrFormat: GPUTextureFormat
|
|
2601
|
+
depthFormat: GPUTextureFormat
|
|
2602
|
+
reversedZ: boolean
|
|
2603
|
+
ids: boolean
|
|
2604
|
+
sampleCount: number
|
|
2605
|
+
presentationFormat: GPUTextureFormat
|
|
2606
|
+
features: string[]
|
|
2607
|
+
errors: { message: string; count: number }[]
|
|
2608
|
+
} {
|
|
2609
|
+
return {
|
|
2610
|
+
hdrFormat: this.hdrFormat,
|
|
2611
|
+
depthFormat: this.depthFormat,
|
|
2612
|
+
reversedZ: this.reversedZ,
|
|
2613
|
+
ids: mrtIdsEnabled(),
|
|
2614
|
+
sampleCount: Engine.MULTISAMPLE_COUNT,
|
|
2615
|
+
presentationFormat: this.presentationFormat,
|
|
2616
|
+
features: this.device ? [...this.device.features].sort() : [],
|
|
2617
|
+
errors: [...this.gpuErrors].map(([message, count]) => ({ message, count })),
|
|
2618
|
+
}
|
|
2619
|
+
}
|
|
2620
|
+
|
|
2433
2621
|
private rebuildCompositeBindGroup(): void {
|
|
2434
2622
|
if (!this.device || !this.hdrResolveTexture || !this.compositeBloomView || !this.depthReadView) return
|
|
2435
2623
|
if (!this.castBuffer) return
|
|
@@ -3091,6 +3279,11 @@ export class Engine {
|
|
|
3091
3279
|
paramsData,
|
|
3092
3280
|
hasBackground,
|
|
3093
3281
|
hasForeground,
|
|
3282
|
+
// The author's OWN source, not the assembled module: the assembled one
|
|
3283
|
+
// always carries the accessors (as real readers or as the zero stubs),
|
|
3284
|
+
// so matching against it would report every effect as a reader and the
|
|
3285
|
+
// attachment would be stored exactly as often as before.
|
|
3286
|
+
readsIds: /\brz(?:ObjectAt|MaterialAt)\s*\(/.test(wgsl),
|
|
3094
3287
|
anchors,
|
|
3095
3288
|
// The effect's own clock starts now. Per effect so that one installed
|
|
3096
3289
|
// later still gets a frame where rzGridFrame() is 0 and can seed.
|
|
@@ -3690,7 +3883,10 @@ export class Engine {
|
|
|
3690
3883
|
return {
|
|
3691
3884
|
ok: true,
|
|
3692
3885
|
state: {
|
|
3693
|
-
|
|
3886
|
+
// Ribbons declared by this effect. The INSTANCE count is no longer
|
|
3887
|
+
// baked here — it follows the live subject count and is computed per
|
|
3888
|
+
// draw (see drawTrails).
|
|
3889
|
+
slots,
|
|
3694
3890
|
uniform,
|
|
3695
3891
|
data: new Float32Array(4),
|
|
3696
3892
|
pipeline,
|
|
@@ -3752,13 +3948,27 @@ export class Engine {
|
|
|
3752
3948
|
// The clock upload happens once, on the camera draw: queue writes land
|
|
3753
3949
|
// before the encoder submits, so both passes read the same value — the
|
|
3754
3950
|
// mirror draw writing it again would only write it twice.
|
|
3951
|
+
// Instances follow the LIVE subject count, not MAX_EFFECT_SUBJECTS.
|
|
3952
|
+
//
|
|
3953
|
+
// This used to be baked at install as slots x 4 x (samples-1) x subs, so a
|
|
3954
|
+
// scene with ONE character issued four characters' worth of ribbon quads
|
|
3955
|
+
// and threw three quarters of them away as degenerate — every frame, at
|
|
3956
|
+
// every sample length. Vertex invocations with no fragments are cheap, not
|
|
3957
|
+
// free, and they scale with the sample count, which is what made a longer
|
|
3958
|
+
// trail expensive.
|
|
3959
|
+
//
|
|
3960
|
+
// The shader decodes [ribbon][subject][segment] with the same number out
|
|
3961
|
+
// of its uniform, so the two cannot drift: change one without the other
|
|
3962
|
+
// and ribbons land on the wrong subject rather than merely costing more.
|
|
3963
|
+
const live = Math.max(1, this.castSubjectCount)
|
|
3755
3964
|
if (view === "camera") {
|
|
3756
3965
|
t.data[0] = this.sceneClock - e.epochScene
|
|
3966
|
+
t.data[1] = live
|
|
3757
3967
|
this.device.queue.writeBuffer(t.uniform, 0, t.data.buffer as ArrayBuffer)
|
|
3758
3968
|
}
|
|
3759
3969
|
pass.setPipeline(t.pipeline)
|
|
3760
3970
|
pass.setBindGroup(0, view === "mirror" ? t.mirrorBind : t.bind)
|
|
3761
|
-
pass.draw(6, t.
|
|
3971
|
+
pass.draw(6, t.slots * live * (TRAIL_SAMPLES - 1) * TRAIL_SUBDIVISIONS)
|
|
3762
3972
|
}
|
|
3763
3973
|
}
|
|
3764
3974
|
|
|
@@ -4187,6 +4397,26 @@ export class Engine {
|
|
|
4187
4397
|
throw new Error("WebGPU is not supported in this browser.")
|
|
4188
4398
|
}
|
|
4189
4399
|
this.device = device
|
|
4400
|
+
// Every validation error this device ever raises, kept.
|
|
4401
|
+
//
|
|
4402
|
+
// WebGPU does not throw for a bad pipeline: createRenderPipeline hands back
|
|
4403
|
+
// an object that is already invalid, and the complaint arrives here instead
|
|
4404
|
+
// — or nowhere, if nobody is listening. Nobody was. That is why a device
|
|
4405
|
+
// that disagrees with this engine has, until now, had no way to say so: the
|
|
4406
|
+
// pipeline is built, setPipeline poisons the pass that uses it, and the
|
|
4407
|
+
// symptom reaches the user as geometry that is simply absent, with a clean
|
|
4408
|
+
// console. A browser is not obliged to agree with Dawn about what is legal,
|
|
4409
|
+
// and the two places this engine knowingly leans on Dawn's reading are both
|
|
4410
|
+
// in the scene pass (see scene-contract's writeMask-0 note).
|
|
4411
|
+
//
|
|
4412
|
+
// Bounded, and not on the console by default: a pass that fails validation
|
|
4413
|
+
// fails it again every frame, so an unbounded log is a memory leak with a
|
|
4414
|
+
// frame counter and an unconditional console.error is a browser tab that
|
|
4415
|
+
// stops responding. First N distinct messages, counted thereafter.
|
|
4416
|
+
device.addEventListener("uncapturederror", (e) => {
|
|
4417
|
+
const message = (e as GPUUncapturedErrorEvent).error.message
|
|
4418
|
+
this.noteGpuError(message)
|
|
4419
|
+
})
|
|
4190
4420
|
if (hasRg11b10) this.hdrFormat = "rg11b10ufloat"
|
|
4191
4421
|
// The override has the last word, including over a device that would have
|
|
4192
4422
|
// been left on the fallback anyway — asking for the format you are already
|
|
@@ -4252,6 +4482,15 @@ export class Engine {
|
|
|
4252
4482
|
this.createPipelines()
|
|
4253
4483
|
this.setupResize()
|
|
4254
4484
|
Engine.instance = this
|
|
4485
|
+
// One line, at init, naming the three answers that differ between two
|
|
4486
|
+
// browsers on the same machine. Not a debug flag and not a readout — it is
|
|
4487
|
+
// the identity of the renderer that was actually built, and on a device that
|
|
4488
|
+
// cannot be attached to a debugger it is the only way to know which of the
|
|
4489
|
+
// three paths is running. Every graphics application prints this.
|
|
4490
|
+
const r = this.gpuReport()
|
|
4491
|
+
console.info(
|
|
4492
|
+
`[reze] hdr=${r.hdrFormat} depth=${r.depthFormat} reversedZ=${r.reversedZ} ids=${r.ids} msaa=${r.sampleCount}`,
|
|
4493
|
+
)
|
|
4255
4494
|
}
|
|
4256
4495
|
|
|
4257
4496
|
// One-shot bake of EEVEE's combined BRDF LUT — DFG (bsdf_lut_frag.glsl) packed
|
|
@@ -4260,6 +4499,45 @@ export class Engine {
|
|
|
4260
4499
|
// .ba = LTC magnitude → ltc_brdf_scale_from_lut
|
|
4261
4500
|
// One texture fetch per fragment replaces the previous 2–3 taps. rgba8unorm
|
|
4262
4501
|
// (vs rgba16float) halves sample bandwidth; DFG/LTC values fit [0,1] cleanly.
|
|
4502
|
+
/** The frost tile the ground samples instead of evaluating fbm per pixel. */
|
|
4503
|
+
private groundNoiseTexture!: GPUTexture
|
|
4504
|
+
private groundNoiseView!: GPUTextureView
|
|
4505
|
+
|
|
4506
|
+
/**
|
|
4507
|
+
* Bake the ground's frost noise once — the same fbm the shader used to run
|
|
4508
|
+
* per pixel, rendered to a seamless 1024² r8unorm tile at init.
|
|
4509
|
+
*
|
|
4510
|
+
* Why this exists is measured, not argued: on WebKit the ground's whole cost
|
|
4511
|
+
* was this evaluation (see the note at the sample site in ground.ts). The
|
|
4512
|
+
* bake is one fullscreen pass at init — under a millisecond, once — and the
|
|
4513
|
+
* per-pixel cost becomes a single level-0 texture read.
|
|
4514
|
+
*/
|
|
4515
|
+
private bakeGroundNoise() {
|
|
4516
|
+
this.groundNoiseTexture = this.device.createTexture({
|
|
4517
|
+
label: "ground frost noise (baked)",
|
|
4518
|
+
size: [GROUND_NOISE_SIZE, GROUND_NOISE_SIZE],
|
|
4519
|
+
format: "r8unorm",
|
|
4520
|
+
usage: GPUTextureUsage.RENDER_ATTACHMENT | GPUTextureUsage.TEXTURE_BINDING,
|
|
4521
|
+
})
|
|
4522
|
+
this.groundNoiseView = this.groundNoiseTexture.createView()
|
|
4523
|
+
const module = this.device.createShaderModule({ label: "ground noise bake", code: GROUND_NOISE_BAKE_WGSL })
|
|
4524
|
+
const pipeline = this.device.createRenderPipeline({
|
|
4525
|
+
label: "ground noise bake",
|
|
4526
|
+
layout: "auto",
|
|
4527
|
+
vertex: { module, entryPoint: "vs" },
|
|
4528
|
+
fragment: { module, entryPoint: "fs", targets: [{ format: "r8unorm" }] },
|
|
4529
|
+
primitive: { topology: "triangle-list" },
|
|
4530
|
+
})
|
|
4531
|
+
const encoder = this.device.createCommandEncoder({ label: "ground noise bake" })
|
|
4532
|
+
const pass = encoder.beginRenderPass({
|
|
4533
|
+
colorAttachments: [{ view: this.groundNoiseView, loadOp: "clear", storeOp: "store" }],
|
|
4534
|
+
})
|
|
4535
|
+
pass.setPipeline(pipeline)
|
|
4536
|
+
pass.draw(3)
|
|
4537
|
+
pass.end()
|
|
4538
|
+
this.device.queue.submit([encoder.finish()])
|
|
4539
|
+
}
|
|
4540
|
+
|
|
4263
4541
|
private bakeBrdfLut() {
|
|
4264
4542
|
if (BRDF_LUT_SIZE !== LTC_MAG_LUT_SIZE) {
|
|
4265
4543
|
throw new Error("BRDF LUT bake requires DFG size == LTC size (both 64).")
|
|
@@ -4798,23 +5076,62 @@ export class Engine {
|
|
|
4798
5076
|
// occluded behind it. Color targets kept for pass compatibility, writeMask 0.
|
|
4799
5077
|
const prepassModule = this.device.createShaderModule({
|
|
4800
5078
|
label: "transparent depth prepass",
|
|
4801
|
-
code:
|
|
5079
|
+
code: transparentDepthPrepassWgsl(),
|
|
4802
5080
|
})
|
|
4803
|
-
|
|
4804
|
-
label: "transparent depth prepass",
|
|
5081
|
+
const prepassDesc = {
|
|
4805
5082
|
layout: mainPipelineLayout,
|
|
4806
5083
|
vertex: { module: prepassModule, entryPoint: "vs", buffers: fullVertexBuffers as GPUVertexBufferLayout[] },
|
|
5084
|
+
primitive: { cullMode: "none" as GPUCullMode },
|
|
5085
|
+
multisample: { count: Engine.MULTISAMPLE_COUNT },
|
|
5086
|
+
depthStencil: {
|
|
5087
|
+
format: this.depthFormat,
|
|
5088
|
+
depthWriteEnabled: true,
|
|
5089
|
+
depthCompare: this.depthAhead,
|
|
5090
|
+
},
|
|
5091
|
+
}
|
|
5092
|
+
this.depthPrepassPipeline = this.device.createRenderPipeline({
|
|
5093
|
+
label: "opaque depth prepass",
|
|
5094
|
+
...prepassDesc,
|
|
4807
5095
|
fragment: {
|
|
4808
5096
|
module: prepassModule,
|
|
4809
5097
|
entryPoint: "fs",
|
|
4810
5098
|
targets: sceneTargetsFor("depth-prepass", this.sceneFormats),
|
|
4811
5099
|
},
|
|
4812
|
-
|
|
4813
|
-
|
|
5100
|
+
})
|
|
5101
|
+
// The SOLID prime: same module, cutoff forced to exactly 1.0. Only texels
|
|
5102
|
+
// whose blend ignores the destination may pre-claim depth in the
|
|
5103
|
+
// transparent phase — see the override's note in depth-prepass.ts.
|
|
5104
|
+
this.solidPrepassPipeline = this.device.createRenderPipeline({
|
|
5105
|
+
label: "transparent solid prepass",
|
|
5106
|
+
...prepassDesc,
|
|
5107
|
+
fragment: {
|
|
5108
|
+
module: prepassModule,
|
|
5109
|
+
entryPoint: "fs",
|
|
5110
|
+
constants: { CUTOFF: 1.0 },
|
|
5111
|
+
targets: sceneTargetsFor("depth-prepass", this.sceneFormats),
|
|
5112
|
+
},
|
|
5113
|
+
})
|
|
5114
|
+
// The HAIR prime: solid texels only, and stencil-fenced off the eye
|
|
5115
|
+
// silhouette. It records after the non-hair opaque draws, so the eye has
|
|
5116
|
+
// already written its stencil — not-equal here is what keeps the primed
|
|
5117
|
+
// hair depth from ever claiming the pixels the see-through-hair pass needs
|
|
5118
|
+
// the eye to survive on. (Bundle draws use the PASS's stencil reference;
|
|
5119
|
+
// only pipeline/bind/vertex state resets across executeBundles.)
|
|
5120
|
+
this.hairPrimePipeline = this.device.createRenderPipeline({
|
|
5121
|
+
label: "hair depth prime",
|
|
5122
|
+
...prepassDesc,
|
|
4814
5123
|
depthStencil: {
|
|
4815
|
-
|
|
4816
|
-
|
|
4817
|
-
|
|
5124
|
+
...prepassDesc.depthStencil,
|
|
5125
|
+
stencilFront: { compare: "not-equal", failOp: "keep", depthFailOp: "keep", passOp: "keep" },
|
|
5126
|
+
stencilBack: { compare: "not-equal", failOp: "keep", depthFailOp: "keep", passOp: "keep" },
|
|
5127
|
+
stencilReadMask: 0xff,
|
|
5128
|
+
stencilWriteMask: 0,
|
|
5129
|
+
},
|
|
5130
|
+
fragment: {
|
|
5131
|
+
module: prepassModule,
|
|
5132
|
+
entryPoint: "fs",
|
|
5133
|
+
constants: { CUTOFF: 1.0 },
|
|
5134
|
+
targets: sceneTargetsFor("depth-prepass", this.sceneFormats),
|
|
4818
5135
|
},
|
|
4819
5136
|
})
|
|
4820
5137
|
|
|
@@ -4852,7 +5169,7 @@ export class Engine {
|
|
|
4852
5169
|
fragment: { module: shadowShader, entryPoint: "fs", targets: [] },
|
|
4853
5170
|
primitive: { cullMode: "none" },
|
|
4854
5171
|
depthStencil: {
|
|
4855
|
-
format:
|
|
5172
|
+
format: Engine.SHADOW_DEPTH_FORMAT,
|
|
4856
5173
|
depthWriteEnabled: true,
|
|
4857
5174
|
depthCompare: "less-equal",
|
|
4858
5175
|
// The shadow map keeps the NON-reversed convention (orthographicLh maps
|
|
@@ -4874,7 +5191,7 @@ export class Engine {
|
|
|
4874
5191
|
this.device.createTexture({
|
|
4875
5192
|
label: `shadow map cascade ${i}`,
|
|
4876
5193
|
size: [c.mapSize, c.mapSize],
|
|
4877
|
-
format:
|
|
5194
|
+
format: Engine.SHADOW_DEPTH_FORMAT,
|
|
4878
5195
|
usage: GPUTextureUsage.RENDER_ATTACHMENT | GPUTextureUsage.TEXTURE_BINDING,
|
|
4879
5196
|
}),
|
|
4880
5197
|
)
|
|
@@ -4882,6 +5199,7 @@ export class Engine {
|
|
|
4882
5199
|
|
|
4883
5200
|
// One-shot bake of Blender EEVEE's combined BRDF LUT (DFG + LTC packed rgba8unorm).
|
|
4884
5201
|
this.bakeBrdfLut()
|
|
5202
|
+
this.bakeGroundNoise()
|
|
4885
5203
|
this.agxFallbackTexture = this.device.createTexture({
|
|
4886
5204
|
label: "AgX LUT fallback",
|
|
4887
5205
|
size: [1, 1, 1],
|
|
@@ -4974,6 +5292,9 @@ export class Engine {
|
|
|
4974
5292
|
{ binding: 9, visibility: GPUShaderStage.FRAGMENT, texture: { sampleType: "float" } },
|
|
4975
5293
|
{ binding: 10, visibility: GPUShaderStage.FRAGMENT, sampler: {} },
|
|
4976
5294
|
{ binding: 11, visibility: GPUShaderStage.FRAGMENT, texture: { sampleType: "depth", multisampled: true } },
|
|
5295
|
+
// The baked frost tile — see bakeGroundNoise. Sampled with binding 10's
|
|
5296
|
+
// repeat sampler, so it brings no sampler of its own.
|
|
5297
|
+
{ binding: 12, visibility: GPUShaderStage.FRAGMENT, texture: { sampleType: "float" } },
|
|
4977
5298
|
],
|
|
4978
5299
|
})
|
|
4979
5300
|
const groundShadowShader = this.device.createShaderModule({
|
|
@@ -5030,7 +5351,7 @@ export class Engine {
|
|
|
5030
5351
|
|
|
5031
5352
|
const outlineShaderModule = this.device.createShaderModule({
|
|
5032
5353
|
label: "outline shaders",
|
|
5033
|
-
code:
|
|
5354
|
+
code: outlineShaderWgsl(),
|
|
5034
5355
|
})
|
|
5035
5356
|
|
|
5036
5357
|
this.outlinePipeline = this.createRenderPipeline({
|
|
@@ -5518,6 +5839,21 @@ export class Engine {
|
|
|
5518
5839
|
}
|
|
5519
5840
|
|
|
5520
5841
|
private handleResize() {
|
|
5842
|
+
// No device, nothing to size.
|
|
5843
|
+
//
|
|
5844
|
+
// Three callers reach this, and two of them can arrive before init() has a
|
|
5845
|
+
// device or after teardown has released one: setRenderSize is PUBLIC and
|
|
5846
|
+
// unordered with respect to init, and the ResizeObserver keeps firing across
|
|
5847
|
+
// a hot reload while the replaced engine is still mounted. Both landed on
|
|
5848
|
+
// `this.device.createTexture` and threw — which is why this only shows up
|
|
5849
|
+
// during development, and why 0.43 never saw it: setRenderSize did not exist
|
|
5850
|
+
// to be called early.
|
|
5851
|
+
//
|
|
5852
|
+
// Returning is correct rather than merely quiet. fixedRenderSize has already
|
|
5853
|
+
// been recorded by the time we get here, and init() ends with its own
|
|
5854
|
+
// handleResize — so the size asked for before the device existed is applied
|
|
5855
|
+
// in full the moment there is something to apply it to.
|
|
5856
|
+
if (!this.device) return
|
|
5521
5857
|
// Fixed override (offline/video rendering) wins; otherwise track CSS size × dpr.
|
|
5522
5858
|
const dpr = window.devicePixelRatio || 1
|
|
5523
5859
|
const width = this.fixedRenderSize ? this.fixedRenderSize.width : Math.floor(this.canvas.clientWidth * dpr)
|
|
@@ -6836,13 +7172,25 @@ export class Engine {
|
|
|
6836
7172
|
return key
|
|
6837
7173
|
}
|
|
6838
7174
|
|
|
7175
|
+
/** True while a stage is in the scene. Two things turn on it: the built-in
|
|
7176
|
+
* ground plane must not draw, and the far shadow cascade has nothing to
|
|
7177
|
+
* cover without one (see the cascade loop). */
|
|
7178
|
+
hasStage(): boolean {
|
|
7179
|
+
for (const inst of this.modelInstances.values()) if (inst.isStage) return true
|
|
7180
|
+
return false
|
|
7181
|
+
}
|
|
7182
|
+
|
|
6839
7183
|
/** True while a stage is in the scene, which is when the built-in ground plane
|
|
6840
7184
|
* must not draw. */
|
|
6841
7185
|
groundIsSuppressed(): boolean {
|
|
6842
|
-
|
|
6843
|
-
return false
|
|
7186
|
+
return this.hasStage()
|
|
6844
7187
|
}
|
|
6845
7188
|
|
|
7189
|
+
/** Per cascade: does its map currently hold nothing but the cleared far plane?
|
|
7190
|
+
* Set by the cascade loop, which skips a cascade that is unwanted and already
|
|
7191
|
+
* cleared rather than re-clearing it every frame. */
|
|
7192
|
+
private shadowCascadeCleared: boolean[] = []
|
|
7193
|
+
|
|
6846
7194
|
removeModel(name: string): void {
|
|
6847
7195
|
const inst = this.modelInstances.get(name)
|
|
6848
7196
|
if (!inst) return
|
|
@@ -7205,7 +7553,7 @@ export class Engine {
|
|
|
7205
7553
|
const gm = inst.gpuMorph
|
|
7206
7554
|
if (!gm || !gm.dispatchNeeded) continue
|
|
7207
7555
|
if (!pass) {
|
|
7208
|
-
pass = encoder.beginComputePass({ label: "morph compute" })
|
|
7556
|
+
pass = encoder.beginComputePass({ label: "morph compute", timestampWrites: this.stamps("morph") })
|
|
7209
7557
|
pass.setPipeline(this.morphComputePipeline)
|
|
7210
7558
|
}
|
|
7211
7559
|
pass.setBindGroup(0, gm.bindGroup)
|
|
@@ -7458,6 +7806,111 @@ export class Engine {
|
|
|
7458
7806
|
flags[o + 20] = f
|
|
7459
7807
|
}
|
|
7460
7808
|
if (this.cullModelBuffer) this.device.queue.writeBuffer(this.cullModelBuffer, 0, data.buffer as ArrayBuffer)
|
|
7809
|
+
this.updateCasterSphere(data)
|
|
7810
|
+
}
|
|
7811
|
+
|
|
7812
|
+
/**
|
|
7813
|
+
* One sphere containing every shadow caster in the scene, for the ground.
|
|
7814
|
+
*
|
|
7815
|
+
* The ground's PCF is the most expensive thing in the frame on a tile-based
|
|
7816
|
+
* GPU — nine hardware-bilinear comparisons per pixel on a full-coverage draw,
|
|
7817
|
+
* which is what 0.33.2 was about and what a second cascade quietly undid. But
|
|
7818
|
+
* the floor is vastly larger than the thing standing on it, and a pixel the
|
|
7819
|
+
* character cannot possibly shadow does not need to ask the shadow map: the
|
|
7820
|
+
* answer is lit, and nine taps is an expensive way to spell it.
|
|
7821
|
+
*
|
|
7822
|
+
* So the ground gets a bound and tests against it in ALU. This reuses the
|
|
7823
|
+
* spheres the cull already builds every frame — an AABB over POSED bone
|
|
7824
|
+
* positions grown by the skin margin, which its own note calls a bound rather
|
|
7825
|
+
* than an estimate, so a jump or a physics-driven skirt is inside it by
|
|
7826
|
+
* construction. Union, not per model: one sphere is one test, and the ground
|
|
7827
|
+
* shader must not loop over the cast.
|
|
7828
|
+
*
|
|
7829
|
+
* A RIGID caster (a stage) leaves its cull sphere zeroed deliberately — the
|
|
7830
|
+
* cull reads its boxes instead — so any rigid model disables this entirely by
|
|
7831
|
+
* setting radius to -1. Wrong here is a missing shadow, and a scene with a
|
|
7832
|
+
* stage keeps the taps rather than risk one.
|
|
7833
|
+
*/
|
|
7834
|
+
private updateCasterSphere(data: Float32Array): void {
|
|
7835
|
+
const out = this.casterSphere
|
|
7836
|
+
out[3] = 0
|
|
7837
|
+
let cx = 0
|
|
7838
|
+
let cy = 0
|
|
7839
|
+
let cz = 0
|
|
7840
|
+
let r = 0
|
|
7841
|
+
let any = false
|
|
7842
|
+
for (let i = 0; i < this.cullModels.length; i++) {
|
|
7843
|
+
const inst = this.cullModels[i]
|
|
7844
|
+
if (!inst.model.visible || inst.shadowDrawCalls.length === 0) continue
|
|
7845
|
+
if (inst.rigid) {
|
|
7846
|
+
// No sphere to read. Bail out of the whole optimisation.
|
|
7847
|
+
out[3] = -1
|
|
7848
|
+
return
|
|
7849
|
+
}
|
|
7850
|
+
const o = i * Engine.CULL_MODEL_FLOATS + 16
|
|
7851
|
+
const x = data[o]
|
|
7852
|
+
const y = data[o + 1]
|
|
7853
|
+
const z = data[o + 2]
|
|
7854
|
+
const rad = data[o + 3]
|
|
7855
|
+
if (rad <= 0) continue
|
|
7856
|
+
if (!any) {
|
|
7857
|
+
cx = x
|
|
7858
|
+
cy = y
|
|
7859
|
+
cz = z
|
|
7860
|
+
r = rad
|
|
7861
|
+
any = true
|
|
7862
|
+
continue
|
|
7863
|
+
}
|
|
7864
|
+
// Union of two spheres, the standard construction: if one already contains
|
|
7865
|
+
// the other keep it, else grow along the line between the centres.
|
|
7866
|
+
const dx = x - cx
|
|
7867
|
+
const dy = y - cy
|
|
7868
|
+
const dz = z - cz
|
|
7869
|
+
const d = Math.hypot(dx, dy, dz)
|
|
7870
|
+
if (d + rad <= r) continue
|
|
7871
|
+
if (d + r <= rad) {
|
|
7872
|
+
cx = x
|
|
7873
|
+
cy = y
|
|
7874
|
+
cz = z
|
|
7875
|
+
r = rad
|
|
7876
|
+
continue
|
|
7877
|
+
}
|
|
7878
|
+
const nr = (d + r + rad) * 0.5
|
|
7879
|
+
const t = (nr - r) / d
|
|
7880
|
+
cx += dx * t
|
|
7881
|
+
cy += dy * t
|
|
7882
|
+
cz += dz * t
|
|
7883
|
+
r = nr
|
|
7884
|
+
}
|
|
7885
|
+
out[0] = cx
|
|
7886
|
+
out[1] = cy
|
|
7887
|
+
out[2] = cz
|
|
7888
|
+
out[3] = any ? r : 0
|
|
7889
|
+
}
|
|
7890
|
+
|
|
7891
|
+
/** Every shadow caster in one sphere: (x, y, z, radius). radius 0 = nothing
|
|
7892
|
+
* casts, -1 = do not use (a rigid caster has no sphere). See updateCasterSphere. */
|
|
7893
|
+
private casterSphere = new Float32Array(4)
|
|
7894
|
+
|
|
7895
|
+
/** The ground's uniform block, kept so the caster sphere can be refreshed in
|
|
7896
|
+
* it every frame rather than rebuilding the buffer (addGround allocates). */
|
|
7897
|
+
private groundMaterialData: Float32Array | null = null
|
|
7898
|
+
|
|
7899
|
+
/**
|
|
7900
|
+
* Push this frame's caster sphere into the ground's uniform.
|
|
7901
|
+
*
|
|
7902
|
+
* Four floats, one writeBuffer, and only while a ground exists. Rebuilding the
|
|
7903
|
+
* block the way addGround does would allocate a buffer and a bind group per
|
|
7904
|
+
* frame, which is the cost this is trying to remove rather than a way to pay
|
|
7905
|
+
* it somewhere else.
|
|
7906
|
+
*/
|
|
7907
|
+
private writeGroundCasterSphere(): void {
|
|
7908
|
+
const gb = this.groundMaterialData
|
|
7909
|
+
if (!gb || !this.groundShadowMaterialBuffer) return
|
|
7910
|
+
if (gb[20] === this.casterSphere[0] && gb[21] === this.casterSphere[1] &&
|
|
7911
|
+
gb[22] === this.casterSphere[2] && gb[23] === this.casterSphere[3]) return
|
|
7912
|
+
gb.set(this.casterSphere, 20)
|
|
7913
|
+
this.device.queue.writeBuffer(this.groundShadowMaterialBuffer, 80, this.casterSphere as Float32Array<ArrayBuffer>)
|
|
7461
7914
|
}
|
|
7462
7915
|
|
|
7463
7916
|
/**
|
|
@@ -7594,7 +8047,6 @@ export class Engine {
|
|
|
7594
8047
|
}
|
|
7595
8048
|
if (this.modelInstances.size === 0) {
|
|
7596
8049
|
this.opaqueBundle = null
|
|
7597
|
-
this.transparentBundle = null
|
|
7598
8050
|
this.mirrorOpaqueBundle = null
|
|
7599
8051
|
this.mirrorTransparentBundle = null
|
|
7600
8052
|
this.shadowBundles = []
|
|
@@ -7606,14 +8058,15 @@ export class Engine {
|
|
|
7606
8058
|
this.forEachInstance((inst) => this.renderModelOpaquePhase(opaque, inst, camView))
|
|
7607
8059
|
this.opaqueBundle = opaque.finish({ label: "opaque phase" })
|
|
7608
8060
|
|
|
7609
|
-
|
|
7610
|
-
|
|
7611
|
-
|
|
7612
|
-
|
|
7613
|
-
//
|
|
7614
|
-
//
|
|
7615
|
-
//
|
|
7616
|
-
//
|
|
8061
|
+
// NO camera transparent bundle. The camera pass draws that phase directly —
|
|
8062
|
+
// see the note at the executeBundles call for what recording one cost on
|
|
8063
|
+
// WebKit. Recording it anyway "in case" is not free and not harmless: it is
|
|
8064
|
+
// work on every rebuild, and a live bundle beside a direct draw of the same
|
|
8065
|
+
// phase is an invitation to execute it again.
|
|
8066
|
+
//
|
|
8067
|
+
// The MIRROR pair below keeps both bundles, and is allowed to: that pass
|
|
8068
|
+
// hands them to a single executeBundles with nothing direct in between,
|
|
8069
|
+
// which is the pattern that works.
|
|
7617
8070
|
const mirrorView = this.sceneView("mirror")
|
|
7618
8071
|
const mo = this.device.createRenderBundleEncoder({ label: "mirror opaque phase", ...scene })
|
|
7619
8072
|
this.forEachInstance((inst) => this.renderModelOpaquePhase(mo, inst, mirrorView))
|
|
@@ -7631,7 +8084,7 @@ export class Engine {
|
|
|
7631
8084
|
const shadow = this.device.createRenderBundleEncoder({
|
|
7632
8085
|
label: `shadow pass, cascade ${ci}`,
|
|
7633
8086
|
colorFormats: [],
|
|
7634
|
-
depthStencilFormat:
|
|
8087
|
+
depthStencilFormat: Engine.SHADOW_DEPTH_FORMAT,
|
|
7635
8088
|
})
|
|
7636
8089
|
shadow.setPipeline(this.shadowDepthPipeline)
|
|
7637
8090
|
this.forEachInstance((inst) => this.drawInstanceShadow(shadow, inst, ci))
|
|
@@ -7648,6 +8101,25 @@ export class Engine {
|
|
|
7648
8101
|
return { querySet: this.timestampQuerySet, beginningOfPassWriteIndex: i * 2, endOfPassWriteIndex: i * 2 + 1 }
|
|
7649
8102
|
}
|
|
7650
8103
|
|
|
8104
|
+
/**
|
|
8105
|
+
* Half a stamp, for a component that is several passes rather than one.
|
|
8106
|
+
*
|
|
8107
|
+
* Bloom is nine render passes — a prefilter blit, a downsample chain and an
|
|
8108
|
+
* upsample chain — and what anyone wants to know is what the PYRAMID cost, not
|
|
8109
|
+
* what its fourth mip cost. Both fields of GPURenderPassTimestampWrites are
|
|
8110
|
+
* optional, so the opening query goes on the first pass and the closing one on
|
|
8111
|
+
* the last, and the pair reads as one span across everything between.
|
|
8112
|
+
*/
|
|
8113
|
+
private stampOpen(pass: (typeof Engine.TIMED_PASSES)[number]): GPURenderPassTimestampWrites | undefined {
|
|
8114
|
+
if (!this.timestampQuerySet) return undefined
|
|
8115
|
+
return { querySet: this.timestampQuerySet, beginningOfPassWriteIndex: Engine.TIMED_PASSES.indexOf(pass) * 2 }
|
|
8116
|
+
}
|
|
8117
|
+
|
|
8118
|
+
private stampClose(pass: (typeof Engine.TIMED_PASSES)[number]): GPURenderPassTimestampWrites | undefined {
|
|
8119
|
+
if (!this.timestampQuerySet) return undefined
|
|
8120
|
+
return { querySet: this.timestampQuerySet, endOfPassWriteIndex: Engine.TIMED_PASSES.indexOf(pass) * 2 + 1 }
|
|
8121
|
+
}
|
|
8122
|
+
|
|
7651
8123
|
/**
|
|
7652
8124
|
* Resolve this frame's timings and start a readback, at most one in flight.
|
|
7653
8125
|
*
|
|
@@ -7659,6 +8131,8 @@ export class Engine {
|
|
|
7659
8131
|
private resolveTimestamps(encoder: GPUCommandEncoder): void {
|
|
7660
8132
|
const qs = this.timestampQuerySet
|
|
7661
8133
|
if (!qs || !this.timestampResolve || !this.timestampRead) return
|
|
8134
|
+
// Nobody has asked. See getGpuTimings — the read is what enrols.
|
|
8135
|
+
if (!this.timestampsWanted) return
|
|
7662
8136
|
const count = Engine.TIMED_PASSES.length * 2
|
|
7663
8137
|
encoder.resolveQuerySet(qs, 0, count, this.timestampResolve, 0)
|
|
7664
8138
|
if (this.timestampBusy) return
|
|
@@ -7698,11 +8172,27 @@ export class Engine {
|
|
|
7698
8172
|
* The regression guard for the draw-path work: these are the numbers that say
|
|
7699
8173
|
* whether restructuring cost anything, which is the claim being made — not
|
|
7700
8174
|
* whether it made the scene faster, which was never the goal.
|
|
8175
|
+
*
|
|
8176
|
+
* ASKING IS WHAT TURNS IT ON. The first call to this enrols the engine in the
|
|
8177
|
+
* per-frame readback; until then resolveTimestamps does nothing. That is why
|
|
8178
|
+
* the first call returns null even on a device that can measure — the answer
|
|
8179
|
+
* arrives a frame or two later, which is already true of these numbers and
|
|
8180
|
+
* documented on resolveTimestamps.
|
|
8181
|
+
*
|
|
8182
|
+
* The alternative was what this used to do: resolve the query set, copy it to
|
|
8183
|
+
* a staging buffer and map that buffer, every frame, on every device, for a
|
|
8184
|
+
* reader that in this codebase did not exist. A map is a synchronisation point
|
|
8185
|
+
* and the whole path is instrumentation — paying for it unasked is the same
|
|
8186
|
+
* mistake as shipping a debug flag, only invisible.
|
|
7701
8187
|
*/
|
|
7702
8188
|
getGpuTimings(): Record<string, number> | null {
|
|
8189
|
+
this.timestampsWanted = true
|
|
7703
8190
|
return this.gpuPassMs
|
|
7704
8191
|
}
|
|
7705
8192
|
|
|
8193
|
+
/** Set by the first getGpuTimings() call. See it for why asking is the switch. */
|
|
8194
|
+
private timestampsWanted = false
|
|
8195
|
+
|
|
7706
8196
|
private dispatchCull(encoder: GPUCommandEncoder): void {
|
|
7707
8197
|
if (this.cullListDirty) this.rebuildCullList()
|
|
7708
8198
|
if (!this.cullBindGroup || this.cullDraws.length === 0) return
|
|
@@ -8297,7 +8787,8 @@ export class Engine {
|
|
|
8297
8787
|
// Shadow map is already created in setupPipelines()
|
|
8298
8788
|
// 20 floats: 16 for the original block, then (mirrorBlur, pad, pad, pad)
|
|
8299
8789
|
// keeping the uniform vec4-aligned.
|
|
8300
|
-
const gb = new Float32Array(
|
|
8790
|
+
const gb = new Float32Array(24)
|
|
8791
|
+
this.groundMaterialData = gb
|
|
8301
8792
|
gb[0] = diffuseColor.x
|
|
8302
8793
|
gb[1] = diffuseColor.y
|
|
8303
8794
|
gb[2] = diffuseColor.z
|
|
@@ -8317,6 +8808,23 @@ export class Engine {
|
|
|
8317
8808
|
this.groundMirror = gb[15]
|
|
8318
8809
|
gb[16] = Math.min(Math.max(mirrorBlur, 0), 1)
|
|
8319
8810
|
this.groundMirrorBlur = gb[16]
|
|
8811
|
+
// gb[17] — does the FAR cascade hold anything?
|
|
8812
|
+
//
|
|
8813
|
+
// It holds something only when a stage is loaded; that is what it exists for
|
|
8814
|
+
// and the cascade loop already skips drawing into it otherwise, leaving it
|
|
8815
|
+
// cleared. A cleared depth map compares as "no occluder", so the ground's far
|
|
8816
|
+
// branch is nine comparison taps whose answer is known in advance.
|
|
8817
|
+
//
|
|
8818
|
+
// That branch runs wherever the NEAR cascade does not reach, and the near one
|
|
8819
|
+
// is a 64-unit box around the camera target — so on a floor receding to the
|
|
8820
|
+
// horizon it is most of the visible pixels, on the most expensive
|
|
8821
|
+
// full-coverage draw in the frame. Skipping it is free in the exact sense:
|
|
8822
|
+
// the shader takes vis = 1.0, which is what the taps would have returned.
|
|
8823
|
+
gb[17] = this.hasStage() ? 1 : 0
|
|
8824
|
+
// gb[20..23] — the caster sphere, refreshed every frame by
|
|
8825
|
+
// writeGroundCasterSphere. Zero here so a frame that renders before the
|
|
8826
|
+
// first cull (there is one) reads "nothing casts" and skips the taps, which
|
|
8827
|
+
// is true: no model has been posed yet.
|
|
8320
8828
|
this.groundShadowMaterialBuffer = this.device.createBuffer({
|
|
8321
8829
|
size: gb.byteLength,
|
|
8322
8830
|
usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST,
|
|
@@ -8351,6 +8859,7 @@ export class Engine {
|
|
|
8351
8859
|
{ binding: 9, resource: this.mirrorColorView! },
|
|
8352
8860
|
{ binding: 10, resource: this.materialSampler },
|
|
8353
8861
|
{ binding: 11, resource: this.mirrorDepthReadView! },
|
|
8862
|
+
{ binding: 12, resource: this.groundNoiseView },
|
|
8354
8863
|
],
|
|
8355
8864
|
})
|
|
8356
8865
|
if (this.groundDrawCall) this.groundDrawCall.bindGroup = this.groundShadowBindGroup
|
|
@@ -8873,7 +9382,19 @@ export class Engine {
|
|
|
8873
9382
|
|
|
8874
9383
|
// CPU alpha sampler for sheerness classification (see textureAlphaCache).
|
|
8875
9384
|
// Canvas 2D premultiplies RGB on readback, but the ALPHA channel is exact.
|
|
8876
|
-
|
|
9385
|
+
const alphaPlane = buildAlphaSampler(source, rgba, width, height)
|
|
9386
|
+
// Loud, because the fallback is WRONG rather than merely absent: a material
|
|
9387
|
+
// with no alpha plane scores avg 1 / translucentFrac 0, which routes sheer
|
|
9388
|
+
// fabric into the OPAQUE bucket and changes what the frame looks like. A
|
|
9389
|
+
// readback that fails is therefore a rendering bug, not a missing nicety,
|
|
9390
|
+
// and it must not reach the user as "the dress looks different on my phone".
|
|
9391
|
+
if (!alphaPlane) {
|
|
9392
|
+
console.warn(
|
|
9393
|
+
`[reze] alpha readback failed for ${cacheKey} — this material will be classified OPAQUE, ` +
|
|
9394
|
+
`so sheer fabric will not blend. The canvas 2D readback is what failed.`,
|
|
9395
|
+
)
|
|
9396
|
+
}
|
|
9397
|
+
this.textureAlphaCache.set(cacheKey, alphaPlane)
|
|
8877
9398
|
|
|
8878
9399
|
const mipLevelCount = Math.floor(Math.log2(Math.max(width, height))) + 1
|
|
8879
9400
|
const texture = this.device.createTexture({
|
|
@@ -9634,11 +10155,50 @@ export class Engine {
|
|
|
9634
10155
|
const dofOn = this.depthOfField.enabled
|
|
9635
10156
|
// ANY effect: one foreground mount anywhere in the scene, or one ribbon,
|
|
9636
10157
|
// is enough to make the pass store its depth instead of discarding it.
|
|
9637
|
-
|
|
9638
|
-
|
|
10158
|
+
// Ribbons are NOT in this list, and removing them is the single largest
|
|
10159
|
+
// bandwidth saving in the frame on a tile-based GPU.
|
|
10160
|
+
//
|
|
10161
|
+
// They were, from when a ribbon was its own layer drawn after the scene and
|
|
10162
|
+
// depth-tested BY HAND against the stored buffer. That layer is gone —
|
|
10163
|
+
// ribbons draw inside this pass and the hardware depth test replaced what
|
|
10164
|
+
// they read it for (see trails.ts, "Binding 3 is GONE"). The clause outlived
|
|
10165
|
+
// the change by about twelve hours and then sat here.
|
|
10166
|
+
//
|
|
10167
|
+
// What it cost: this flag decides whether the pass STORES its depth or
|
|
10168
|
+
// discards it into tile memory, and the buffer is depth32float-stencil8 at
|
|
10169
|
+
// the pass's sample count — on a retina canvas that is a nine-figure number
|
|
10170
|
+
// of bytes written to RAM every frame, for a texture nothing then sampled.
|
|
10171
|
+
// Chrome hides it (an immediate-mode GPU has depth in memory regardless);
|
|
10172
|
+
// Apple's TBDR does not, which is exactly the reported shape: adding a hand
|
|
10173
|
+
// ribbon costs a lot of fps on Safari and almost nothing on Chrome.
|
|
10174
|
+
//
|
|
10175
|
+
// The two real readers are both in the composite and both have their own
|
|
10176
|
+
// flag above: linearDepth() feeds the DoF gather and the depth handed to a
|
|
10177
|
+
// foreground mount. Nothing else binds depthTex at all.
|
|
10178
|
+
const depthRead = dofOn || this.effects.some((e) => e.hasForeground)
|
|
9639
10179
|
this.renderPassDescriptor.depthStencilAttachment!.depthStoreOp = depthRead ? "store" : "discard"
|
|
9640
10180
|
if (depthRead) this.writeDepthOfFieldUniforms()
|
|
9641
10181
|
|
|
10182
|
+
// The id attachment, on exactly the same terms as the depth above it.
|
|
10183
|
+
//
|
|
10184
|
+
// It is the most expensive STORE in the pass — rg16uint at the pass's sample
|
|
10185
|
+
// count, ~33MB a frame at 1080p — and a uint target cannot be resolved, so
|
|
10186
|
+
// storing is the only way to get it out. It was stored unconditionally, for
|
|
10187
|
+
// every scene, whether or not anything read it. Nothing usually does: the
|
|
10188
|
+
// readers are rzObjectAt / rzMaterialAt in an effect that masks itself to one
|
|
10189
|
+
// character, and the id-buffer debug view.
|
|
10190
|
+
//
|
|
10191
|
+
// Discarding is not the same as removing. Every pipeline still declares the
|
|
10192
|
+
// attachment and the pass still carries it, so nothing is rebuilt and no
|
|
10193
|
+
// shader changes — the frame is bit-identical either way, because the only
|
|
10194
|
+
// difference is whether tile memory is written back to RAM after a pass
|
|
10195
|
+
// whose result no one is going to read.
|
|
10196
|
+
const idAtt = (this.renderPassDescriptor.colorAttachments as GPURenderPassColorAttachment[])[2]
|
|
10197
|
+
if (idAtt) {
|
|
10198
|
+
const idsRead = this.idDebug || this.effects.some((e) => e.readsIds)
|
|
10199
|
+
idAtt.storeOp = idsRead ? "store" : "discard"
|
|
10200
|
+
}
|
|
10201
|
+
|
|
9642
10202
|
const encoder = this.device.createCommandEncoder()
|
|
9643
10203
|
|
|
9644
10204
|
// GPU vertex morphs: write morphed positions into vertex buffers before any pass reads
|
|
@@ -9649,6 +10209,8 @@ export class Engine {
|
|
|
9649
10209
|
// are settled, before the passes that draw from them.
|
|
9650
10210
|
if (this.reflectionActive) this.updateMirrorCamera()
|
|
9651
10211
|
if (hasModels) this.dispatchCull(encoder)
|
|
10212
|
+
// After the cull, which is what recomputes the spheres it unions.
|
|
10213
|
+
this.writeGroundCasterSphere()
|
|
9652
10214
|
|
|
9653
10215
|
// After the cull, because a rebuild there can reallocate the argument
|
|
9654
10216
|
// buffers and a bundle captures the buffer it recorded against.
|
|
@@ -9660,7 +10222,25 @@ export class Engine {
|
|
|
9660
10222
|
// keeps PCF-sampling a character that is no longer in the scene. One clearing
|
|
9661
10223
|
// pass on the transition to empty, then it stops.
|
|
9662
10224
|
if (hasModels || this.shadowMapPopulated) {
|
|
10225
|
+
// The far cascade is the STAGE cascade, and it costs a full pass over the
|
|
10226
|
+
// whole cast every frame to say so. Its own spec explains what it is for —
|
|
10227
|
+
// "a set piece 100 units out still throws" — and a scene with no stage has
|
|
10228
|
+
// no set piece: every caster sits inside the near cascade's 64-unit box,
|
|
10229
|
+
// which follows the camera target, and the far map's only readers are
|
|
10230
|
+
// ground pixels beyond that box, where nothing is casting.
|
|
10231
|
+
//
|
|
10232
|
+
// So when no stage is loaded it is drawn ONCE, cleared, and then skipped —
|
|
10233
|
+
// the same shape as shadowMapPopulated above, and for the same reason. A
|
|
10234
|
+
// cleared depth map reads as "no occluder", which is the correct answer
|
|
10235
|
+
// here rather than a missing one. Load a stage and it comes straight back.
|
|
10236
|
+
//
|
|
10237
|
+
// 0.43 had ONE shadow map. This is half of what the second one costs.
|
|
10238
|
+
const stage = this.hasStage()
|
|
9663
10239
|
for (let ci = 0; ci < SHADOW_CASCADES.length; ci++) {
|
|
10240
|
+
const wanted = ci === 0 || stage
|
|
10241
|
+
// Already cleared and still unwanted — nothing to do, and the map still
|
|
10242
|
+
// holds the far plane from the pass that cleared it.
|
|
10243
|
+
if (!wanted && this.shadowCascadeCleared[ci]) continue
|
|
9664
10244
|
const sp = encoder.beginRenderPass({
|
|
9665
10245
|
// One timestamp pair exists for "shadow"; the near cascade wears it.
|
|
9666
10246
|
timestampWrites: ci === 0 ? this.stamps("shadow") : undefined,
|
|
@@ -9676,8 +10256,9 @@ export class Engine {
|
|
|
9676
10256
|
// per-frame boolean, and baking it into a bundle would make toggling a
|
|
9677
10257
|
// model re-record. It lives in the cull compute now, which zeroes the
|
|
9678
10258
|
// instance count of an invisible model's draws.
|
|
9679
|
-
if (this.shadowBundles[ci]) sp.executeBundles([this.shadowBundles[ci]])
|
|
10259
|
+
if (wanted && this.shadowBundles[ci]) sp.executeBundles([this.shadowBundles[ci]])
|
|
9680
10260
|
sp.end()
|
|
10261
|
+
this.shadowCascadeCleared[ci] = !wanted
|
|
9681
10262
|
}
|
|
9682
10263
|
this.shadowMapPopulated = hasModels
|
|
9683
10264
|
}
|
|
@@ -9708,8 +10289,44 @@ export class Engine {
|
|
|
9708
10289
|
// anyway — eye writes it, hair tests not-equal, hairOverEyes tests equal.
|
|
9709
10290
|
pass.setStencilReference(Engine.STENCIL_EYE_VALUE)
|
|
9710
10291
|
if (this.opaqueBundle) pass.executeBundles([this.opaqueBundle])
|
|
10292
|
+
// Re-asserted after the bundle, not merely set once before it.
|
|
10293
|
+
//
|
|
10294
|
+
// Stencil reference is pass state a bundle cannot carry — GPURenderBundleEncoder
|
|
10295
|
+
// has no setStencilReference — which is why it was hoisted above the bundle in
|
|
10296
|
+
// the first place. But "cannot carry" and "cannot disturb" are different
|
|
10297
|
+
// claims, and only the first is specified. Everything below this line that
|
|
10298
|
+
// stencil-tests (hair at not-equal, outline hulls at not-equal) reads a
|
|
10299
|
+
// reference of 0 instead of 1 if a replay resets it, and not-equal against 0
|
|
10300
|
+
// is FALSE for the cleared buffer — every such fragment silently rejected.
|
|
10301
|
+
// One redundant word against a whole class of invisible failure.
|
|
10302
|
+
pass.setStencilReference(Engine.STENCIL_EYE_VALUE)
|
|
9711
10303
|
if (this.hasGround) this.renderGround(pass)
|
|
9712
|
-
|
|
10304
|
+
// The transparent phase is drawn DIRECTLY, and must stay that way. It is the
|
|
10305
|
+
// one part of this pass that is not bundled, so the reason is worth keeping.
|
|
10306
|
+
//
|
|
10307
|
+
// It WAS a bundle, and on WebKit the entire transparent bucket vanished while
|
|
10308
|
+
// the opaque bucket and the ground rendered perfectly — sheer fabric simply
|
|
10309
|
+
// absent, with no validation error anywhere. It was not the fragments: with
|
|
10310
|
+
// alpha forced to 1 they still never appeared, the cull reported every draw
|
|
10311
|
+
// visible with its GPU and CPU halves agreeing, and a cast model's
|
|
10312
|
+
// transparent draws use the SAME pipeline, bind groups and depth state as its
|
|
10313
|
+
// opaque ones (pipelineForDrawCall, forceDepthWrite). Identical draws,
|
|
10314
|
+
// identical state, one bucket rendering.
|
|
10315
|
+
//
|
|
10316
|
+
// What differed was only how they reached the pass: the opaque bundle is the
|
|
10317
|
+
// FIRST executeBundles here, and the transparent one was the SECOND, issued
|
|
10318
|
+
// after direct commands (the ground). Legal, and correct on Dawn. Not
|
|
10319
|
+
// replayed on WebKit. The mirror pass is the counter-example that pins the
|
|
10320
|
+
// shape of it — it passes BOTH bundles to a single executeBundles with
|
|
10321
|
+
// nothing direct in between, and has never lost a draw.
|
|
10322
|
+
//
|
|
10323
|
+
// So the rule this pass now keeps: at most one executeBundles, and nothing
|
|
10324
|
+
// direct before it. Bundling this phase again means first moving the ground
|
|
10325
|
+
// into the opaque bundle so the two can go in one call, the way the mirror
|
|
10326
|
+
// does it. The saving that buys is CPU encode time over a handful of draws,
|
|
10327
|
+
// which was never this renderer's bottleneck.
|
|
10328
|
+
const camView = this.sceneView("camera")
|
|
10329
|
+
this.forEachInstance((inst) => this.renderModelTransparentPhase(pass, inst, camView))
|
|
9713
10330
|
// Last in the pass: depth-tested against everything drawn above, so a
|
|
9714
10331
|
// particle behind the character is simply hidden, and still inside the HDR
|
|
9715
10332
|
// target so an `@bloom` effect reaches the pyramid below.
|
|
@@ -9730,11 +10347,18 @@ export class Engine {
|
|
|
9730
10347
|
// 3. Upsample (top-down): bloomUp[N-2] = tent(bloomDown[N-1]) + bloomDown[N-2],
|
|
9731
10348
|
// then bloomUp[i] = tent(bloomUp[i+1]) + bloomDown[i] until i=0 (9-tap tent)
|
|
9732
10349
|
// Composite reads bloomUp[0] and adds tint * intensity * bloom before Filmic.
|
|
9733
|
-
|
|
10350
|
+
// bloomContributes() gates the whole pyramid, not just its intensity. The
|
|
10351
|
+
// composite still SAMPLES bloomUp[0] unconditionally, which is safe and
|
|
10352
|
+
// deliberate: it scales what it reads by the same effective intensity, so a
|
|
10353
|
+
// stale or never-written pyramid is multiplied by zero. Skipping the build
|
|
10354
|
+
// is therefore invisible in the frame and nine render passes cheaper.
|
|
10355
|
+
if (this.bloomContributes() && this.bloomBlitBindGroup && this.compositeBindGroup && this.bloomMipCount > 0) {
|
|
9734
10356
|
const bloomAtt = this.bloomPassDescriptor.colorAttachments as GPURenderPassColorAttachment[]
|
|
9735
10357
|
|
|
9736
|
-
// 1. Blit
|
|
10358
|
+
// 1. Blit — opens the pyramid's timing span. See stampOpen: the nine
|
|
10359
|
+
// passes below read as ONE component, which is the only useful grain.
|
|
9737
10360
|
bloomAtt[0].view = this.bloomDownMipViews[0]
|
|
10361
|
+
this.bloomPassDescriptor.timestampWrites = this.stampOpen("bloom")
|
|
9738
10362
|
const pBlit = encoder.beginRenderPass(this.bloomPassDescriptor)
|
|
9739
10363
|
pBlit.setPipeline(this.bloomBlitPipeline)
|
|
9740
10364
|
pBlit.setBindGroup(0, this.bloomBlitBindGroup)
|
|
@@ -9742,6 +10366,7 @@ export class Engine {
|
|
|
9742
10366
|
pBlit.end()
|
|
9743
10367
|
|
|
9744
10368
|
// 2. Downsample chain
|
|
10369
|
+
this.bloomPassDescriptor.timestampWrites = undefined
|
|
9745
10370
|
for (let i = 1; i < this.bloomMipCount; i++) {
|
|
9746
10371
|
bloomAtt[0].view = this.bloomDownMipViews[i]
|
|
9747
10372
|
const p = encoder.beginRenderPass(this.bloomPassDescriptor)
|
|
@@ -9757,6 +10382,8 @@ export class Engine {
|
|
|
9757
10382
|
for (let k = 0; k < upSteps; k++) {
|
|
9758
10383
|
const levelIdx = topIdx - k // writes bloomUp[levelIdx]
|
|
9759
10384
|
bloomAtt[0].view = this.bloomUpMipViews[levelIdx]
|
|
10385
|
+
// The LAST upsample closes the span opened on the blit.
|
|
10386
|
+
this.bloomPassDescriptor.timestampWrites = k === upSteps - 1 ? this.stampClose("bloom") : undefined
|
|
9760
10387
|
const p = encoder.beginRenderPass(this.bloomPassDescriptor)
|
|
9761
10388
|
p.setPipeline(this.bloomUpsamplePipeline)
|
|
9762
10389
|
p.setBindGroup(0, this.bloomUpsampleBindGroups[k])
|
|
@@ -10325,16 +10952,29 @@ export class Engine {
|
|
|
10325
10952
|
* makes outlines compose like MMD: every material drawn later in the author's
|
|
10326
10953
|
* order covers earlier hulls, and each hull sits over everything drawn before it.
|
|
10327
10954
|
*/
|
|
10955
|
+
/** Is this draw's compiled class "hair"? Ungrouped draws never are — the
|
|
10956
|
+
* neutral pipeline is the auto class. */
|
|
10957
|
+
private isHairDraw(inst: ModelInstance, dc: DrawCall): boolean {
|
|
10958
|
+
if (!dc.groupId) return false
|
|
10959
|
+
const install = inst.styleGroups.get(dc.groupId)
|
|
10960
|
+
return install?.renderClass === "hair"
|
|
10961
|
+
}
|
|
10962
|
+
|
|
10328
10963
|
private drawMaterials(
|
|
10329
10964
|
pass: GPURenderPassEncoder | GPURenderBundleEncoder,
|
|
10330
10965
|
inst: ModelInstance,
|
|
10331
10966
|
type: "opaque" | "transparent",
|
|
10332
10967
|
view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
|
|
10968
|
+
// The opaque phase walks its author order twice — non-hair, then hair — so
|
|
10969
|
+
// the hair depth prime can sit between the eye's stencil write and the hair
|
|
10970
|
+
// colour that must respect it. See renderModelOpaquePhase.
|
|
10971
|
+
only?: "hair" | "non-hair",
|
|
10333
10972
|
): void {
|
|
10334
10973
|
let currentPipeline: GPURenderPipeline | null = null
|
|
10335
10974
|
let bound = false
|
|
10336
10975
|
for (const draw of inst.drawCalls) {
|
|
10337
10976
|
if (draw.type !== type) continue
|
|
10977
|
+
if (only && (only === "hair") !== this.isHairDraw(inst, draw)) continue
|
|
10338
10978
|
if (!bound) {
|
|
10339
10979
|
pass.setBindGroup(0, view.perFrame)
|
|
10340
10980
|
pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
|
|
@@ -10405,16 +11045,155 @@ export class Engine {
|
|
|
10405
11045
|
view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
|
|
10406
11046
|
): void {
|
|
10407
11047
|
this.setModelDrawState(pass, inst)
|
|
10408
|
-
|
|
11048
|
+
// Depth first, colour second — the close-up fix, and the oldest one there
|
|
11049
|
+
// is. See drawOpaqueDepthPrepass.
|
|
11050
|
+
this.drawOpaqueDepthPrepass(pass, inst, view)
|
|
11051
|
+
// The opaque author order, in two walks with the hair prime between them.
|
|
11052
|
+
//
|
|
11053
|
+
// Hair could not join the plain prepass: primed hair depth would depth-
|
|
11054
|
+
// reject the eye before it writes the stencil the see-through-hair pass
|
|
11055
|
+
// needs. But the trick only needs the eye BEFORE hair, not before
|
|
11056
|
+
// everything — so the non-hair walk runs first (the eye writes stencil
|
|
11057
|
+
// against real face depth, exactly as it always did), the prime then lays
|
|
11058
|
+
// hair depth down stencil-fenced off the eye silhouette, and the hair walk
|
|
11059
|
+
// shades once per pixel instead of once per card.
|
|
11060
|
+
//
|
|
11061
|
+
// The one thing this reorders: hair now draws after any opaque material
|
|
11062
|
+
// authored later than it. A soft hair edge over such a material blends
|
|
11063
|
+
// over the material instead of over whatever the framebuffer held mid-
|
|
11064
|
+
// order — deterministic where it used to be accidental, and only at
|
|
11065
|
+
// sub-alpha edge texels over late-authored geometry.
|
|
11066
|
+
this.drawMaterials(pass, inst, "opaque", view, "non-hair")
|
|
11067
|
+
this.drawHairDepthPrime(pass, inst, view)
|
|
11068
|
+
this.drawMaterials(pass, inst, "opaque", view, "hair")
|
|
10409
11069
|
this.drawHairOverEyes(pass, inst, view)
|
|
10410
11070
|
}
|
|
10411
11071
|
|
|
11072
|
+
/** Depth-only prime of the hair's alpha-1 texels, stencil-fenced off the eye
|
|
11073
|
+
* silhouette. See the note at its call site and hairPrimePipeline. */
|
|
11074
|
+
private drawHairDepthPrime(
|
|
11075
|
+
pass: GPURenderPassEncoder | GPURenderBundleEncoder,
|
|
11076
|
+
inst: ModelInstance,
|
|
11077
|
+
view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
|
|
11078
|
+
): void {
|
|
11079
|
+
let bound = false
|
|
11080
|
+
for (const draw of inst.drawCalls) {
|
|
11081
|
+
if (draw.type !== "opaque" || !this.isHairDraw(inst, draw)) continue
|
|
11082
|
+
if (!bound) {
|
|
11083
|
+
pass.setPipeline(this.hairPrimePipeline)
|
|
11084
|
+
pass.setBindGroup(0, view.perFrame)
|
|
11085
|
+
pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
|
|
11086
|
+
bound = true
|
|
11087
|
+
}
|
|
11088
|
+
pass.setBindGroup(2, draw.bindGroup)
|
|
11089
|
+
this.issueDraw(pass, draw, view.args)
|
|
11090
|
+
}
|
|
11091
|
+
}
|
|
11092
|
+
|
|
11093
|
+
/**
|
|
11094
|
+
* Depth-only prime of the plain opaque draws, so each covered pixel SHADES
|
|
11095
|
+
* once instead of once per layer.
|
|
11096
|
+
*
|
|
11097
|
+
* The oldest fps complaint this engine has — zoom close and the frame drops,
|
|
11098
|
+
* in every material generation back to the earliest — was never the vertices
|
|
11099
|
+
* and never one shader's fault: with the fragment shaders flattened to a
|
|
11100
|
+
* constant the close-up ran smooth with identical geometry, overdraw and
|
|
11101
|
+
* MSAA. The cost is per-fragment shading TIMES how many times a pixel runs
|
|
11102
|
+
* it, and an MMD model at close-up is layers all the way down: cloth over
|
|
11103
|
+
* body, sleeves over cloth, hair over everything. Author-order drawing
|
|
11104
|
+
* shades every layer and then buries all but one.
|
|
11105
|
+
*
|
|
11106
|
+
* So the plain opaque draws lay their depth down first, through the same
|
|
11107
|
+
* depth-only pipeline the transparent bucket keeps for its dormant prepass —
|
|
11108
|
+
* same skinned vertex path (position marked @invariant in both modules, so
|
|
11109
|
+
* the colour pass lands on exactly these depths and its less-equal test
|
|
11110
|
+
* keeps the visible surface and rejects the buried ones), same alpha-0.5
|
|
11111
|
+
* cutout, writeMask 0 on every colour target. The pixels are identical by
|
|
11112
|
+
* construction: this pass writes no colour, and the colour pass draws
|
|
11113
|
+
* exactly what it always drew minus the fragments something opaque provably
|
|
11114
|
+
* covers.
|
|
11115
|
+
*
|
|
11116
|
+
* WHO IS IN. Only render-class "auto" with alpha-mode "opaque" — the body,
|
|
11117
|
+
* face and cloth materials that are the bulk of every model — plus every
|
|
11118
|
+
* ungrouped material (the neutral pipeline is that same class). WHO IS OUT,
|
|
11119
|
+
* each for a reason that would change pixels: EYE front-culls and gates on a
|
|
11120
|
+
* bone read, and pre-filled hair depth over the socket would depth-reject
|
|
11121
|
+
* the eye before it could write the stencil the see-through-hair pass needs
|
|
11122
|
+
* — which is also why HAIR stays out entirely. HASHED alpha (stockings)
|
|
11123
|
+
* discards by a position hash this pass does not run, so priming it would
|
|
11124
|
+
* punch its cutout into the depth buffer at the wrong texels. They all still
|
|
11125
|
+
* BENEFIT: their fragments early-z against the primed depth of whatever
|
|
11126
|
+
* plain opaque surface sits in front of them.
|
|
11127
|
+
*/
|
|
11128
|
+
private drawOpaqueDepthPrepass(
|
|
11129
|
+
pass: GPURenderPassEncoder | GPURenderBundleEncoder,
|
|
11130
|
+
inst: ModelInstance,
|
|
11131
|
+
view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
|
|
11132
|
+
): void {
|
|
11133
|
+
let bound = false
|
|
11134
|
+
for (const draw of inst.drawCalls) {
|
|
11135
|
+
if (draw.type !== "opaque") continue
|
|
11136
|
+
if (draw.groupId) {
|
|
11137
|
+
const install = inst.styleGroups.get(draw.groupId)
|
|
11138
|
+
if (install && !(install.renderClass === "auto" && install.alphaMode === "opaque")) continue
|
|
11139
|
+
}
|
|
11140
|
+
if (!bound) {
|
|
11141
|
+
pass.setPipeline(this.depthPrepassPipeline)
|
|
11142
|
+
pass.setBindGroup(0, view.perFrame)
|
|
11143
|
+
pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
|
|
11144
|
+
bound = true
|
|
11145
|
+
}
|
|
11146
|
+
pass.setBindGroup(2, draw.bindGroup)
|
|
11147
|
+
this.issueDraw(pass, draw, view.args)
|
|
11148
|
+
}
|
|
11149
|
+
}
|
|
11150
|
+
|
|
11151
|
+
/**
|
|
11152
|
+
* Depth-only prime of the transparent bucket's FULLY SOLID texels.
|
|
11153
|
+
*
|
|
11154
|
+
* The dress problem. A "transparent" MMD material is mostly weave at alpha
|
|
11155
|
+
* exactly 1 with sheer margins, and its layers draw in author order — so a
|
|
11156
|
+
* close-up skirt shades every buried panel and then covers the work. The
|
|
11157
|
+
* buried SHEER fragments must shade (their blend reads what is behind), but
|
|
11158
|
+
* at alpha 1 over-blending is plain replacement: the destination cannot
|
|
11159
|
+
* matter, so a fragment buried behind an alpha-1 texel contributes nothing.
|
|
11160
|
+
* Priming depth for exactly those texels (CUTOFF 1.0) rejects the buried
|
|
11161
|
+
* work and cannot move a pixel.
|
|
11162
|
+
*
|
|
11163
|
+
* A STAGE's transparent draws are excluded the way their colour path already
|
|
11164
|
+
* is: stage glass deliberately leaves depth alone so rain and particles
|
|
11165
|
+
* survive behind a dome (see pipelineForDrawCall), and a prime would put the
|
|
11166
|
+
* occlusion right back.
|
|
11167
|
+
*/
|
|
11168
|
+
private drawTransparentSolidPrepass(
|
|
11169
|
+
pass: GPURenderPassEncoder | GPURenderBundleEncoder,
|
|
11170
|
+
inst: ModelInstance,
|
|
11171
|
+
view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
|
|
11172
|
+
): void {
|
|
11173
|
+
if (inst.isStage) return
|
|
11174
|
+
let bound = false
|
|
11175
|
+
for (const draw of inst.drawCalls) {
|
|
11176
|
+
if (draw.type !== "transparent") continue
|
|
11177
|
+
if (!bound) {
|
|
11178
|
+
pass.setPipeline(this.solidPrepassPipeline)
|
|
11179
|
+
pass.setBindGroup(0, view.perFrame)
|
|
11180
|
+
pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
|
|
11181
|
+
bound = true
|
|
11182
|
+
}
|
|
11183
|
+
pass.setBindGroup(2, draw.bindGroup)
|
|
11184
|
+
this.issueDraw(pass, draw, view.args)
|
|
11185
|
+
}
|
|
11186
|
+
}
|
|
11187
|
+
|
|
10412
11188
|
private renderModelTransparentPhase(
|
|
10413
11189
|
pass: GPURenderPassEncoder | GPURenderBundleEncoder,
|
|
10414
11190
|
inst: ModelInstance,
|
|
10415
11191
|
view: { perFrame: GPUBindGroup; args: "camera" | "mirror"; outlines: boolean },
|
|
10416
11192
|
): void {
|
|
11193
|
+
// Draw state FIRST — each phase records into its own bundle encoder, and a
|
|
11194
|
+
// bundle starts with nothing bound.
|
|
10417
11195
|
this.setModelDrawState(pass, inst)
|
|
11196
|
+
this.drawTransparentSolidPrepass(pass, inst, view)
|
|
10418
11197
|
// Transparent: babylon-mmd's forceDepthWrite blending — PMX author order
|
|
10419
11198
|
// with depth write ON. The accepted trade-off after trying every variant:
|
|
10420
11199
|
// · depth-write ON (this): a fold hides its far side; rare view-dependent
|
|
@@ -10434,7 +11213,7 @@ export class Engine {
|
|
|
10434
11213
|
for (const draw of inst.drawCalls) {
|
|
10435
11214
|
if (draw.type !== "transparent") continue
|
|
10436
11215
|
if (!bound) {
|
|
10437
|
-
pass.setPipeline(this.
|
|
11216
|
+
pass.setPipeline(this.depthPrepassPipeline)
|
|
10438
11217
|
pass.setBindGroup(0, this.perFrameBindGroup)
|
|
10439
11218
|
pass.setBindGroup(1, inst.mainPerInstanceBindGroup)
|
|
10440
11219
|
bound = true
|
|
@@ -10584,6 +11363,10 @@ export class Engine {
|
|
|
10584
11363
|
n++
|
|
10585
11364
|
})
|
|
10586
11365
|
u[43] = n
|
|
11366
|
+
// The same number the ribbons size their instance count by — see
|
|
11367
|
+
// drawTrails. Recorded rather than recomputed: this loop is the one place
|
|
11368
|
+
// that knows how many subjects the cast actually ended up holding.
|
|
11369
|
+
this.castSubjectCount = n
|
|
10587
11370
|
this.device.queue.writeBuffer(this.compositeUniformBuffer, 0, u)
|
|
10588
11371
|
// Only what an effect declared, and only while one is installed. A scene
|
|
10589
11372
|
// with no effect writes nothing here at all.
|