@bornengine/engine 0.4.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (213) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +231 -0
  3. package/native/android/Cargo.lock +1848 -0
  4. package/native/android/Cargo.toml +24 -0
  5. package/native/android/src/lib.rs +702 -0
  6. package/native/ios/Cargo.lock +1690 -0
  7. package/native/ios/Cargo.toml +32 -0
  8. package/native/ios/src/lib.rs +1267 -0
  9. package/native/linux/Cargo.lock +3279 -0
  10. package/native/linux/Cargo.toml +29 -0
  11. package/native/linux/src/lib.rs +1331 -0
  12. package/native/macos/Cargo.lock +3310 -0
  13. package/native/macos/Cargo.toml +46 -0
  14. package/native/macos/src/lib.rs +1302 -0
  15. package/native/shared/Cargo.lock +1899 -0
  16. package/native/shared/Cargo.toml +62 -0
  17. package/native/shared/assets/default_font.ttf +0 -0
  18. package/native/shared/build.rs +270 -0
  19. package/native/shared/shaders/common/clouds.wgsl +122 -0
  20. package/native/shared/shaders/common/fog.wgsl +16 -0
  21. package/native/shared/shaders/common/foliage_wind.wgsl +98 -0
  22. package/native/shared/shaders/common/imposter.wgsl +112 -0
  23. package/native/shared/shaders/common/pbr.wgsl +186 -0
  24. package/native/shared/shaders/common/shadows.wgsl +186 -0
  25. package/native/shared/shaders/common/sky.wgsl +8 -0
  26. package/native/shared/shaders/common/tonemap.wgsl +25 -0
  27. package/native/shared/shaders/impulse_field.wgsl +57 -0
  28. package/native/shared/shaders/material_abi.wgsl +383 -0
  29. package/native/shared/shaders/materials/test_minimal.wgsl +42 -0
  30. package/native/shared/src/anim_mixer.rs +61 -0
  31. package/native/shared/src/attach.rs +263 -0
  32. package/native/shared/src/audio/decode.rs +123 -0
  33. package/native/shared/src/audio/mod.rs +863 -0
  34. package/native/shared/src/audio/render.rs +892 -0
  35. package/native/shared/src/audio/spsc.rs +156 -0
  36. package/native/shared/src/audio/stream.rs +226 -0
  37. package/native/shared/src/custom_shaders.rs +104 -0
  38. package/native/shared/src/decals.rs +245 -0
  39. package/native/shared/src/drs.rs +211 -0
  40. package/native/shared/src/engine.rs +261 -0
  41. package/native/shared/src/ffi.rs +116 -0
  42. package/native/shared/src/ffi_core/assets.rs +388 -0
  43. package/native/shared/src/ffi_core/audio_ffi.rs +184 -0
  44. package/native/shared/src/ffi_core/draw.rs +334 -0
  45. package/native/shared/src/ffi_core/game_loop.rs +577 -0
  46. package/native/shared/src/ffi_core/input.rs +234 -0
  47. package/native/shared/src/ffi_core/mod.rs +127 -0
  48. package/native/shared/src/ffi_core/models.rs +1154 -0
  49. package/native/shared/src/ffi_core/ragdoll_ffi.rs +261 -0
  50. package/native/shared/src/ffi_core/scene.rs +626 -0
  51. package/native/shared/src/ffi_core/vfx.rs +212 -0
  52. package/native/shared/src/ffi_core/visual.rs +691 -0
  53. package/native/shared/src/frame_callbacks.rs +122 -0
  54. package/native/shared/src/geometry.rs +236 -0
  55. package/native/shared/src/handles.rs +182 -0
  56. package/native/shared/src/input.rs +448 -0
  57. package/native/shared/src/jolt_sys.rs +822 -0
  58. package/native/shared/src/lib.rs +55 -0
  59. package/native/shared/src/models.rs +1093 -0
  60. package/native/shared/src/models_gltf.rs +1280 -0
  61. package/native/shared/src/particles.rs +391 -0
  62. package/native/shared/src/physics_jolt.rs +1908 -0
  63. package/native/shared/src/picking.rs +298 -0
  64. package/native/shared/src/postfx.rs +345 -0
  65. package/native/shared/src/profiler.rs +492 -0
  66. package/native/shared/src/ragdoll.rs +474 -0
  67. package/native/shared/src/renderer/atmosphere_lut.rs +573 -0
  68. package/native/shared/src/renderer/brdf_lut.rs +154 -0
  69. package/native/shared/src/renderer/draw2d.rs +143 -0
  70. package/native/shared/src/renderer/formats.rs +822 -0
  71. package/native/shared/src/renderer/froxel.rs +421 -0
  72. package/native/shared/src/renderer/gi_bake.rs +653 -0
  73. package/native/shared/src/renderer/graph.rs +462 -0
  74. package/native/shared/src/renderer/hiz.rs +269 -0
  75. package/native/shared/src/renderer/hot_reload.rs +390 -0
  76. package/native/shared/src/renderer/impulse_field.rs +456 -0
  77. package/native/shared/src/renderer/lighting.rs +154 -0
  78. package/native/shared/src/renderer/material_instancing.rs +171 -0
  79. package/native/shared/src/renderer/material_pipeline.rs +700 -0
  80. package/native/shared/src/renderer/material_system.rs +1996 -0
  81. package/native/shared/src/renderer/material_system_tests.rs +601 -0
  82. package/native/shared/src/renderer/material_system_wasm.rs +41 -0
  83. package/native/shared/src/renderer/mod.rs +12556 -0
  84. package/native/shared/src/renderer/model_draw.rs +641 -0
  85. package/native/shared/src/renderer/occlusion.rs +429 -0
  86. package/native/shared/src/renderer/planar_pass.rs +593 -0
  87. package/native/shared/src/renderer/planar_reflection.rs +499 -0
  88. package/native/shared/src/renderer/post_pass.rs +249 -0
  89. package/native/shared/src/renderer/postfx_chain.rs +728 -0
  90. package/native/shared/src/renderer/pt_pass.rs +577 -0
  91. package/native/shared/src/renderer/scene_pass.rs +607 -0
  92. package/native/shared/src/renderer/shader_include.rs +205 -0
  93. package/native/shared/src/renderer/shader_library.rs +135 -0
  94. package/native/shared/src/renderer/shaders/ao.rs +570 -0
  95. package/native/shared/src/renderer/shaders/core.rs +1243 -0
  96. package/native/shared/src/renderer/shaders/env.rs +907 -0
  97. package/native/shared/src/renderer/shaders/gi.rs +810 -0
  98. package/native/shared/src/renderer/shaders/mod.rs +19 -0
  99. package/native/shared/src/renderer/shaders/post.rs +1558 -0
  100. package/native/shared/src/renderer/shaders/pt.rs +1859 -0
  101. package/native/shared/src/renderer/shaders/ssgi.rs +1586 -0
  102. package/native/shared/src/renderer/shadow_pass.rs +731 -0
  103. package/native/shared/src/renderer/ssgi_pass.rs +392 -0
  104. package/native/shared/src/renderer/ssr_pass.rs +188 -0
  105. package/native/shared/src/renderer/texture_store.rs +473 -0
  106. package/native/shared/src/renderer/transient.rs +591 -0
  107. package/native/shared/src/renderer/types.rs +941 -0
  108. package/native/shared/src/renderer/util.rs +152 -0
  109. package/native/shared/src/scene.rs +1362 -0
  110. package/native/shared/src/sdf_cache.rs +274 -0
  111. package/native/shared/src/shadows.rs +1036 -0
  112. package/native/shared/src/staging.rs +102 -0
  113. package/native/shared/src/string_header.rs +266 -0
  114. package/native/shared/src/text_renderer.rs +502 -0
  115. package/native/shared/src/textures.rs +197 -0
  116. package/native/tvos/Cargo.lock +1693 -0
  117. package/native/tvos/Cargo.toml +36 -0
  118. package/native/tvos/metal-patched/Cargo.toml +178 -0
  119. package/native/tvos/metal-patched/LICENSE-APACHE +201 -0
  120. package/native/tvos/metal-patched/LICENSE-MIT +25 -0
  121. package/native/tvos/metal-patched/src/acceleration_structure.rs +667 -0
  122. package/native/tvos/metal-patched/src/acceleration_structure_pass.rs +108 -0
  123. package/native/tvos/metal-patched/src/argument.rs +366 -0
  124. package/native/tvos/metal-patched/src/blitpass.rs +102 -0
  125. package/native/tvos/metal-patched/src/buffer.rs +71 -0
  126. package/native/tvos/metal-patched/src/capturedescriptor.rs +76 -0
  127. package/native/tvos/metal-patched/src/capturemanager.rs +113 -0
  128. package/native/tvos/metal-patched/src/commandbuffer.rs +192 -0
  129. package/native/tvos/metal-patched/src/commandqueue.rs +44 -0
  130. package/native/tvos/metal-patched/src/computepass.rs +107 -0
  131. package/native/tvos/metal-patched/src/constants.rs +152 -0
  132. package/native/tvos/metal-patched/src/counters.rs +119 -0
  133. package/native/tvos/metal-patched/src/depthstencil.rs +190 -0
  134. package/native/tvos/metal-patched/src/device.rs +2134 -0
  135. package/native/tvos/metal-patched/src/drawable.rs +39 -0
  136. package/native/tvos/metal-patched/src/encoder.rs +2041 -0
  137. package/native/tvos/metal-patched/src/heap.rs +281 -0
  138. package/native/tvos/metal-patched/src/indirect_encoder.rs +344 -0
  139. package/native/tvos/metal-patched/src/lib.rs +657 -0
  140. package/native/tvos/metal-patched/src/library.rs +902 -0
  141. package/native/tvos/metal-patched/src/mps.rs +575 -0
  142. package/native/tvos/metal-patched/src/pipeline/compute.rs +475 -0
  143. package/native/tvos/metal-patched/src/pipeline/mod.rs +71 -0
  144. package/native/tvos/metal-patched/src/pipeline/render.rs +762 -0
  145. package/native/tvos/metal-patched/src/renderpass.rs +443 -0
  146. package/native/tvos/metal-patched/src/resource.rs +182 -0
  147. package/native/tvos/metal-patched/src/sampler.rs +165 -0
  148. package/native/tvos/metal-patched/src/sync.rs +178 -0
  149. package/native/tvos/metal-patched/src/texture.rs +352 -0
  150. package/native/tvos/metal-patched/src/types.rs +90 -0
  151. package/native/tvos/metal-patched/src/vertexdescriptor.rs +250 -0
  152. package/native/tvos/src/audio_backend.rs +197 -0
  153. package/native/tvos/src/lib.rs +1891 -0
  154. package/native/visionos/Cargo.lock +1693 -0
  155. package/native/visionos/Cargo.toml +40 -0
  156. package/native/visionos/src/audio_backend.rs +197 -0
  157. package/native/visionos/src/lib.rs +1887 -0
  158. package/native/watchos/Cargo.lock +16 -0
  159. package/native/watchos/Cargo.toml +19 -0
  160. package/native/watchos/shaders/bloom_postfx.metal +99 -0
  161. package/native/watchos/src/BloomWatchApp.swift +1267 -0
  162. package/native/watchos/src/BloomWatchAudio.swift +179 -0
  163. package/native/watchos/src/audio.rs +55 -0
  164. package/native/watchos/src/draw_list.rs +229 -0
  165. package/native/watchos/src/ffi_stubs.rs +915 -0
  166. package/native/watchos/src/ffi_stubs_manual.rs +35 -0
  167. package/native/watchos/src/lib.rs +1124 -0
  168. package/native/watchos/src/models.rs +746 -0
  169. package/native/watchos/src/postfx.rs +95 -0
  170. package/native/watchos/src/scene.rs +534 -0
  171. package/native/watchos/src/textures.rs +184 -0
  172. package/native/web/Cargo.lock +1657 -0
  173. package/native/web/Cargo.toml +43 -0
  174. package/native/web/bloom_glue.js +695 -0
  175. package/native/web/build.sh +131 -0
  176. package/native/web/index.html +35 -0
  177. package/native/web/jolt_bridge.js +1519 -0
  178. package/native/web/src/input_ffi.rs +286 -0
  179. package/native/web/src/lib.rs +1796 -0
  180. package/native/web/src/material_ffi.rs +710 -0
  181. package/native/web/src/parity_ffi.rs +343 -0
  182. package/native/web/src/physics_ffi.rs +643 -0
  183. package/native/web/src/ragdoll_ffi.rs +250 -0
  184. package/native/web/src/render_settings.rs +98 -0
  185. package/native/windows/Cargo.lock +1815 -0
  186. package/native/windows/Cargo.toml +68 -0
  187. package/native/windows/src/lib.rs +1486 -0
  188. package/package.json +4279 -0
  189. package/src/audio/index.ts +315 -0
  190. package/src/core/colors.ts +63 -0
  191. package/src/core/index.ts +1206 -0
  192. package/src/core/keys.ts +63 -0
  193. package/src/core/types.ts +104 -0
  194. package/src/index.ts +171 -0
  195. package/src/math/index.ts +516 -0
  196. package/src/mobile/index.ts +294 -0
  197. package/src/models/index.ts +1258 -0
  198. package/src/physics/index.ts +1134 -0
  199. package/src/scene/index.ts +698 -0
  200. package/src/shapes/index.ts +120 -0
  201. package/src/text/index.ts +48 -0
  202. package/src/textures/index.ts +187 -0
  203. package/src/vfx/index.ts +191 -0
  204. package/src/world/index.ts +24 -0
  205. package/src/world/loader.ts +423 -0
  206. package/src/world/prefab.ts +217 -0
  207. package/src/world/render.ts +172 -0
  208. package/src/world/saver.ts +108 -0
  209. package/src/world/serialize.ts +301 -0
  210. package/src/world/terrain.ts +355 -0
  211. package/src/world/types.ts +160 -0
  212. package/src/world/validate.ts +319 -0
  213. package/src/world/version.ts +114 -0
@@ -0,0 +1,1243 @@
1
+ //! Core pipeline shaders: batched 2D, legacy 3D, and the main scene shader (forward MRT).
2
+ //! Split from renderer/shaders.rs.
3
+
4
+ //! WGSL shader strings used by the renderer.
5
+ //!
6
+ //! Pure data — no behavior, no struct definitions. Each `const`
7
+ //! is `pub(super)` so the surrounding `renderer` module (and only
8
+ //! that module) can see it, via `use super::shaders::*;` in
9
+ //! `mod.rs`. Split out so the ~11 500-line renderer file shrinks
10
+ //! to the Rust logic it actually contains.
11
+
12
+ pub(in crate::renderer) const SHADER_2D: &str = "
13
+ struct Uniforms {
14
+ screen_size: vec2<f32>,
15
+ _pad: vec2<f32>,
16
+ view_proj: mat4x4<f32>,
17
+ };
18
+
19
+ struct VertexInput {
20
+ @location(0) position: vec2<f32>,
21
+ @location(1) uv: vec2<f32>,
22
+ @location(2) color: vec4<f32>,
23
+ };
24
+
25
+ struct VertexOutput {
26
+ @builtin(position) clip_position: vec4<f32>,
27
+ @location(0) uv: vec2<f32>,
28
+ @location(1) color: vec4<f32>,
29
+ };
30
+
31
+ @group(0) @binding(0) var<uniform> uniforms: Uniforms;
32
+ @group(1) @binding(0) var tex: texture_2d<f32>;
33
+ @group(1) @binding(1) var tex_sampler: sampler;
34
+
35
+ @vertex
36
+ fn vs_main(in: VertexInput) -> VertexOutput {
37
+ var out: VertexOutput;
38
+ let world_pos = uniforms.view_proj * vec4<f32>(in.position, 0.0, 1.0);
39
+ let ndc_x = (world_pos.x / uniforms.screen_size.x) * 2.0 - 1.0;
40
+ let ndc_y = 1.0 - (world_pos.y / uniforms.screen_size.y) * 2.0;
41
+ out.clip_position = vec4<f32>(ndc_x, ndc_y, 0.0, 1.0);
42
+ out.uv = in.uv;
43
+ out.color = in.color;
44
+ return out;
45
+ }
46
+
47
+ @fragment
48
+ fn fs_main(in: VertexOutput) -> @location(0) vec4<f32> {
49
+ let tex_color = textureSample(tex, tex_sampler, in.uv);
50
+ return tex_color * in.color;
51
+ }
52
+ ";
53
+
54
+ pub(in crate::renderer) const SHADER_3D: &str = "
55
+ struct Uniforms3D {
56
+ mvp: mat4x4<f32>,
57
+ model: mat4x4<f32>,
58
+ prev_mvp: mat4x4<f32>,
59
+ model_tint: vec4<f32>,
60
+ // x = joint-buffer offset, y = skinned flag (cached skinned draws).
61
+ // Always zero on the immediate path — its verts arrive with joint
62
+ // indices pre-offset CPU-side, so vs_main_3d ignores this field.
63
+ misc: vec4<f32>,
64
+ };
65
+
66
+ struct DirLight {
67
+ direction: vec4<f32>,
68
+ color: vec4<f32>,
69
+ };
70
+
71
+ struct PointLight {
72
+ position: vec4<f32>,
73
+ color: vec4<f32>,
74
+ };
75
+
76
+ struct Lighting {
77
+ ambient: vec4<f32>,
78
+ light_dir: vec4<f32>,
79
+ light_color: vec4<f32>,
80
+ dir_light_count: vec4<f32>,
81
+ dir_lights: array<DirLight, 8>,
82
+ point_light_count: vec4<f32>,
83
+ point_lights: array<PointLight, 256>,
84
+ };
85
+
86
+ struct JointMatrices {
87
+ matrices: array<mat4x4<f32>, 1024>,
88
+ };
89
+
90
+ struct VertexInput3D {
91
+ @location(0) position: vec3<f32>,
92
+ @location(1) normal: vec3<f32>,
93
+ @location(2) color: vec4<f32>,
94
+ @location(3) uv: vec2<f32>,
95
+ @location(4) joints: vec4<f32>,
96
+ @location(5) weights: vec4<f32>,
97
+ };
98
+
99
+ struct VertexOutput3D {
100
+ @builtin(position) clip_position: vec4<f32>,
101
+ @location(0) normal: vec3<f32>,
102
+ @location(1) color: vec4<f32>,
103
+ @location(2) uv: vec2<f32>,
104
+ @location(3) world_pos: vec3<f32>,
105
+ @location(4) curr_clip: vec4<f32>,
106
+ @location(5) prev_clip: vec4<f32>,
107
+ };
108
+
109
+ @group(0) @binding(0) var<uniform> u: Uniforms3D;
110
+ @group(1) @binding(0) var<uniform> lighting: Lighting;
111
+ @group(2) @binding(0) var tex3d: texture_2d<f32>;
112
+ @group(2) @binding(1) var tex3d_sampler: sampler;
113
+ @group(3) @binding(0) var<uniform> joints: JointMatrices;
114
+
115
+ @vertex
116
+ fn vs_main_3d(in: VertexInput3D) -> VertexOutput3D {
117
+ var out: VertexOutput3D;
118
+ let total_weight = in.weights.x + in.weights.y + in.weights.z + in.weights.w;
119
+ var pos = vec4<f32>(in.position, 1.0);
120
+ var norm = vec4<f32>(in.normal, 0.0);
121
+ if (total_weight > 0.01) {
122
+ let j0 = u32(in.joints.x); let j1 = u32(in.joints.y);
123
+ let j2 = u32(in.joints.z); let j3 = u32(in.joints.w);
124
+ let skinned_pos = joints.matrices[j0] * pos * in.weights.x
125
+ + joints.matrices[j1] * pos * in.weights.y
126
+ + joints.matrices[j2] * pos * in.weights.z
127
+ + joints.matrices[j3] * pos * in.weights.w;
128
+ let skinned_norm = joints.matrices[j0] * norm * in.weights.x
129
+ + joints.matrices[j1] * norm * in.weights.y
130
+ + joints.matrices[j2] * norm * in.weights.z
131
+ + joints.matrices[j3] * norm * in.weights.w;
132
+ pos = skinned_pos;
133
+ norm = skinned_norm;
134
+ }
135
+ let curr = u.mvp * pos;
136
+ out.clip_position = curr;
137
+ out.curr_clip = curr;
138
+ out.prev_clip = u.prev_mvp * pos;
139
+ out.normal = normalize((u.model * norm).xyz);
140
+ out.world_pos = (u.model * pos).xyz;
141
+ out.color = in.color * u.model_tint;
142
+ out.uv = in.uv;
143
+ return out;
144
+ }
145
+
146
+ struct Fs3DOut {
147
+ @location(0) color: vec4<f32>,
148
+ @location(1) material: vec2<f32>,
149
+ @location(2) velocity: vec2<f32>,
150
+ @location(3) albedo: vec4<f32>,
151
+ };
152
+
153
+ @fragment
154
+ fn fs_main_3d(in: VertexOutput3D) -> Fs3DOut {
155
+ let n = normalize(in.normal);
156
+
157
+ // Ambient
158
+ var lit = lighting.ambient.rgb * lighting.ambient.a;
159
+
160
+ // Legacy directional light (backward compat)
161
+ let legacy_dir = normalize(lighting.light_dir.xyz);
162
+ let legacy_diffuse = max(dot(n, legacy_dir), 0.0);
163
+ lit += lighting.light_color.rgb * lighting.light_dir.w * legacy_diffuse;
164
+
165
+ // Additional directional lights
166
+ let dir_count = u32(lighting.dir_light_count.x);
167
+ for (var i = 0u; i < dir_count; i++) {
168
+ let dl = lighting.dir_lights[i];
169
+ let dir = normalize(dl.direction.xyz);
170
+ let diff = max(dot(n, dir), 0.0);
171
+ lit += dl.color.rgb * dl.direction.w * diff;
172
+ }
173
+
174
+ // Point lights
175
+ let pt_count = u32(lighting.point_light_count.x);
176
+ for (var i = 0u; i < pt_count; i++) {
177
+ let pl = lighting.point_lights[i];
178
+ let to_light = pl.position.xyz - in.world_pos;
179
+ let dist = length(to_light);
180
+ let range = pl.position.w;
181
+ if (dist < range) {
182
+ let dir = to_light / dist;
183
+ let diff = max(dot(n, dir), 0.0);
184
+ let atten = 1.0 - (dist / range);
185
+ let atten2 = atten * atten;
186
+ lit += pl.color.rgb * pl.color.w * diff * atten2;
187
+ }
188
+ }
189
+
190
+ let tex_color = textureSample(tex3d, tex3d_sampler, in.uv);
191
+ // Per-pixel velocity for motion blur / TAA reprojection.
192
+ let curr_ndc = in.curr_clip.xy / in.curr_clip.w;
193
+ let prev_ndc = in.prev_clip.xy / in.prev_clip.w;
194
+ let vel = (curr_ndc - prev_ndc) * 0.5;
195
+ // Immediate-mode 3D draws (drawCube etc.) aren't PBR — output
196
+ // 0 metallic / 1 roughness so SSR doesn't try to reflect them.
197
+ //
198
+ // Alpha comes from the TINT only. Game textures routinely carry a
199
+ // non-opacity alpha channel (Unvanquished armor packs a gloss mask
200
+ // there), and this batch also renders CPU-skinned characters — the
201
+ // player turned semi-transparent through its gloss mask when texture
202
+ // alpha fed the blend. Deliberate fades still work via tint alpha;
203
+ // untextured effect quads bind the white texture (alpha 1) anyway.
204
+ return Fs3DOut(
205
+ vec4<f32>(tex_color.rgb * in.color.rgb * lit, in.color.a),
206
+ vec2<f32>(0.0, 1.0),
207
+ vel,
208
+ vec4<f32>(0.0),
209
+ );
210
+ }
211
+ ";
212
+
213
+ // The cloud deck (common/clouds.wgsl) is prepended verbatim: this shader is a
214
+ // raw source const and does not run through the material preprocessor. Same
215
+ // file the sky pass and the world materials use, so a cloud shadow crossing
216
+ // the terrain also crosses the trees standing in it — which is the whole
217
+ // reason to share it.
218
+ pub(in crate::renderer) const SCENE_SHADER: &str = concat!(
219
+ include_str!("../../../shaders/common/clouds.wgsl"),
220
+ include_str!("../../../shaders/common/foliage_wind.wgsl"),
221
+ r#"
222
+ struct Uniforms3D {
223
+ mvp: mat4x4<f32>,
224
+ model: mat4x4<f32>,
225
+ prev_mvp: mat4x4<f32>,
226
+ model_tint: vec4<f32>,
227
+ // x = joint-buffer offset for this draw, y = 1.0 for skinned cached
228
+ // draws (vs_main_scene then skins in the VS), zw unused.
229
+ misc: vec4<f32>,
230
+ };
231
+
232
+ struct JointMatrices {
233
+ matrices: array<mat4x4<f32>, 1024>,
234
+ };
235
+
236
+ struct DirLight {
237
+ direction: vec4<f32>,
238
+ color: vec4<f32>,
239
+ };
240
+
241
+ struct PointLight {
242
+ position: vec4<f32>,
243
+ color: vec4<f32>,
244
+ };
245
+
246
+ struct Lighting {
247
+ ambient: vec4<f32>,
248
+ light_dir: vec4<f32>,
249
+ light_color: vec4<f32>,
250
+ dir_light_count: vec4<f32>,
251
+ dir_lights: array<DirLight, 8>,
252
+ point_light_count: vec4<f32>,
253
+ point_lights: array<PointLight, 256>,
254
+ camera_pos: vec4<f32>,
255
+ shadow_cascade_vps: array<mat4x4<f32>, 3>,
256
+ shadow_cascade_splits: vec4<f32>,
257
+ shadow_view_matrix: mat4x4<f32>,
258
+ wind: vec4<f32>, // xy=dir, z=amplitude, w=time (foliage sway)
259
+ cloud: vec4<f32>, // x=shadow strength, y=deck height, z=scale, w=drift m/s
260
+ frame_misc: vec4<f32>, // x=delta_time (prev-frame wind, for motion vectors)
261
+ };
262
+
263
+ struct MaterialFactors {
264
+ metal_rough: vec4<f32>, // x=metallic, y=roughness
265
+ emissive: vec4<f32>, // rgb=emissive factor
266
+ };
267
+
268
+ struct VertexInputScene {
269
+ @location(0) position: vec3<f32>,
270
+ @location(1) normal: vec3<f32>,
271
+ @location(2) color: vec4<f32>,
272
+ @location(3) uv: vec2<f32>,
273
+ @location(4) joints: vec4<f32>,
274
+ @location(5) weights: vec4<f32>,
275
+ @location(6) tangent: vec4<f32>,
276
+ };
277
+
278
+ struct VertexOutputScene {
279
+ // EN-044 — @invariant is load-bearing. The depth prepass and the main pass run
280
+ // the SAME vertex entry point, but through different pipelines: the prepass's
281
+ // fragment stage consumes almost none of the varyings, so the compiler is free
282
+ // to optimise the position maths differently (fma contraction, reassociation)
283
+ // and the two depths stop being bit-identical. The main pass then tests Equal
284
+ // against a depth that is one ulp off, every fragment fails, and the entire
285
+ // forest and the player VANISH — which is exactly what happened, and it looked
286
+ // like a 60 fps win. @invariant forbids that: the position must be computed
287
+ // identically in every pipeline that uses this shader.
288
+ @invariant @builtin(position) clip_position: vec4<f32>,
289
+ @location(0) normal: vec3<f32>,
290
+ @location(1) color: vec4<f32>,
291
+ @location(2) uv: vec2<f32>,
292
+ @location(3) world_pos: vec3<f32>,
293
+ @location(4) tangent: vec4<f32>,
294
+ @location(5) curr_clip: vec4<f32>,
295
+ @location(6) prev_clip: vec4<f32>,
296
+ };
297
+
298
+ @group(0) @binding(0) var<uniform> u: Uniforms3D;
299
+ @group(1) @binding(0) var<uniform> lighting: Lighting;
300
+ @group(1) @binding(1) var env_tex: texture_2d<f32>;
301
+ @group(1) @binding(2) var env_samp: sampler;
302
+ @group(1) @binding(3) var brdf_lut_tex: texture_2d<f32>;
303
+ @group(1) @binding(4) var brdf_lut_samp: sampler;
304
+ @group(1) @binding(5) var shadow_tex_0: texture_depth_2d;
305
+ @group(1) @binding(6) var shadow_tex_1: texture_depth_2d;
306
+ @group(1) @binding(7) var shadow_tex_2: texture_depth_2d;
307
+ @group(1) @binding(8) var shadow_samp: sampler_comparison;
308
+ @group(1) @binding(9) var env_diffuse_tex: texture_2d<f32>;
309
+ @group(2) @binding(0) var base_color_tex: texture_2d<f32>;
310
+ @group(2) @binding(1) var base_color_samp: sampler;
311
+ @group(2) @binding(2) var normal_tex: texture_2d<f32>;
312
+ @group(2) @binding(3) var normal_samp: sampler;
313
+ @group(2) @binding(4) var mr_tex: texture_2d<f32>;
314
+ @group(2) @binding(5) var mr_samp: sampler;
315
+ @group(2) @binding(6) var em_tex: texture_2d<f32>;
316
+ @group(2) @binding(7) var em_samp: sampler;
317
+ @group(2) @binding(8) var<uniform> material: MaterialFactors;
318
+ @group(2) @binding(9) var occ_tex: texture_2d<f32>;
319
+ @group(2) @binding(10) var occ_samp: sampler;
320
+ @group(3) @binding(0) var<uniform> joints: JointMatrices;
321
+ // PT-7 — previous frame's palette, same slot offsets: skinned verts
322
+ // reconstruct last frame's world position from it, giving skeletal
323
+ // motion a REAL velocity (it was exactly zero before).
324
+ @group(3) @binding(1) var<uniform> joints_prev: JointMatrices;
325
+
326
+ const PI: f32 = 3.14159265;
327
+
328
+ fn dir_to_equirect_uv(dir: vec3<f32>) -> vec2<f32> {
329
+ let d = normalize(dir);
330
+ let theta = acos(clamp(d.y, -1.0, 1.0));
331
+ let phi = atan2(d.z, d.x);
332
+ let raw_u = phi / (2.0 * PI);
333
+ let u_coord = raw_u - floor(raw_u);
334
+ let v_coord = theta / PI;
335
+ return vec2<f32>(u_coord, v_coord);
336
+ }
337
+
338
+ // Clamp equirectangular UV so the bilinear filter never reaches
339
+ // across the ±180° seam (u = 0 / 1 boundary). Half a texel on
340
+ // each side keeps every tap on the correct hemisphere.
341
+ fn seamless_equirect_uv(uv: vec2<f32>) -> vec2<f32> {
342
+ let tex_w = f32(textureDimensions(env_tex, 0).x);
343
+ let half_texel = 0.5 / tex_w;
344
+ return vec2<f32>(clamp(uv.x, half_texel, 1.0 - half_texel), uv.y);
345
+ }
346
+
347
+ // Sample the env map at a specific mip level, multiplied by the
348
+ // global env_intensity (lighting.camera_pos.w). Keeps IBL diffuse,
349
+ // IBL specular and the sky pass scaling in sync so loading the same
350
+ // HDR with intensity=2 brightens everything proportionally.
351
+ fn env_sample_lod(dir: vec3<f32>, lod: f32) -> vec3<f32> {
352
+ return textureSampleLevel(env_tex, env_samp, seamless_equirect_uv(dir_to_equirect_uv(dir)), lod).rgb
353
+ * lighting.camera_pos.w;
354
+ }
355
+
356
+ fn env_sample(dir: vec3<f32>) -> vec3<f32> {
357
+ return textureSample(env_tex, env_samp, seamless_equirect_uv(dir_to_equirect_uv(dir))).rgb
358
+ * lighting.camera_pos.w;
359
+ }
360
+
361
+ @vertex
362
+ fn vs_main_scene(in: VertexInputScene) -> VertexOutputScene {
363
+ if (u.misc.y > 0.5) {
364
+ // Skinned draw: u.mvp/u.prev_mvp are the bare view-projection;
365
+ // joint matrices bake world placement for weighted verts, and
366
+ // u.model places the rare rigid (weightless) verts. No wind
367
+ // sway here — characters aren't foliage.
368
+ let total_weight = in.weights.x + in.weights.y + in.weights.z + in.weights.w;
369
+ var world4: vec4<f32>;
370
+ var prev_world4: vec4<f32>;
371
+ var nrm4: vec4<f32>;
372
+ var tan4: vec4<f32>;
373
+ let pos4l = vec4<f32>(in.position, 1.0);
374
+ let nrm4l = vec4<f32>(in.normal, 0.0);
375
+ let tan4l = vec4<f32>(in.tangent.xyz, 0.0);
376
+ if (total_weight > 0.01) {
377
+ // The cached VB keeps RAW joint indices; misc.x is this
378
+ // draw's base slot in the shared 1024-entry joint buffer.
379
+ let j0 = u32(in.joints.x + u.misc.x); let j1 = u32(in.joints.y + u.misc.x);
380
+ let j2 = u32(in.joints.z + u.misc.x); let j3 = u32(in.joints.w + u.misc.x);
381
+ world4 = joints.matrices[j0] * pos4l * in.weights.x
382
+ + joints.matrices[j1] * pos4l * in.weights.y
383
+ + joints.matrices[j2] * pos4l * in.weights.z
384
+ + joints.matrices[j3] * pos4l * in.weights.w;
385
+ // PT-7 — where this vertex WAS: previous palette, same
386
+ // slots. Feeds the velocity MRT so TAA/TSR and the path
387
+ // tracer can reproject skeletal motion.
388
+ prev_world4 = joints_prev.matrices[j0] * pos4l * in.weights.x
389
+ + joints_prev.matrices[j1] * pos4l * in.weights.y
390
+ + joints_prev.matrices[j2] * pos4l * in.weights.z
391
+ + joints_prev.matrices[j3] * pos4l * in.weights.w;
392
+ nrm4 = joints.matrices[j0] * nrm4l * in.weights.x
393
+ + joints.matrices[j1] * nrm4l * in.weights.y
394
+ + joints.matrices[j2] * nrm4l * in.weights.z
395
+ + joints.matrices[j3] * nrm4l * in.weights.w;
396
+ tan4 = joints.matrices[j0] * tan4l * in.weights.x
397
+ + joints.matrices[j1] * tan4l * in.weights.y
398
+ + joints.matrices[j2] * tan4l * in.weights.z
399
+ + joints.matrices[j3] * tan4l * in.weights.w;
400
+ } else {
401
+ world4 = u.model * pos4l;
402
+ prev_world4 = world4;
403
+ nrm4 = u.model * nrm4l;
404
+ tan4 = u.model * tan4l;
405
+ }
406
+ var o: VertexOutputScene;
407
+ let c = u.mvp * world4;
408
+ o.clip_position = c;
409
+ o.curr_clip = c;
410
+ o.prev_clip = u.prev_mvp * prev_world4;
411
+ o.world_pos = world4.xyz;
412
+ o.normal = normalize(nrm4.xyz);
413
+ o.color = in.color * u.model_tint;
414
+ o.uv = in.uv;
415
+ o.tangent = vec4<f32>(normalize(tan4.xyz), in.tangent.w);
416
+ return o;
417
+ }
418
+ var out: VertexOutputScene;
419
+ var local = in.position;
420
+ // Hierarchical foliage wind (common/foliage_wind.wgsl). u.misc.z is the
421
+ // per-draw foliage amount — 0 for everything that is not a plant, so the
422
+ // world does not sway. This replaces a sway that only ever moved ALPHA-CUT
423
+ // materials, which meant leaf cards fluttered and every trunk was rigid.
424
+ //
425
+ // is_leaf comes from the alpha cutoff, so cards get the fast flutter layer
426
+ // and wood does not.
427
+ var prev_local = local;
428
+ if (u.misc.z > 0.0 && lighting.wind.z > 0.0) {
429
+ // is_leaf from the alpha cutoff: cards get the fast flutter layer, wood
430
+ // does not. Same helper the shadow pass calls, so the tree and its shadow
431
+ // bend together.
432
+ let is_leaf = select(0.0, 1.0, material.metal_rough.w > 0.0);
433
+ local = foliage_wind_local(in.position, u.model, lighting.wind, u.misc.z, is_leaf);
434
+ // Last frame's offset too, so TAA gets a real velocity for a moving leaf
435
+ // instead of 0 and stops smearing the canopy into the sky behind it.
436
+ var w_prev = lighting.wind;
437
+ w_prev.w = lighting.wind.w - lighting.frame_misc.x;
438
+ prev_local = foliage_wind_local(in.position, u.model, w_prev, u.misc.z, is_leaf);
439
+ }
440
+ let pos4 = vec4<f32>(local, 1.0);
441
+ let curr = u.mvp * pos4;
442
+ out.clip_position = curr;
443
+ out.curr_clip = curr;
444
+ out.prev_clip = u.prev_mvp * vec4<f32>(prev_local, 1.0);
445
+ let world4 = u.model * pos4;
446
+ out.world_pos = world4.xyz;
447
+ out.normal = normalize((u.model * vec4<f32>(in.normal, 0.0)).xyz);
448
+ out.color = in.color * u.model_tint;
449
+ out.uv = in.uv;
450
+ out.tangent = vec4<f32>(normalize((u.model * vec4<f32>(in.tangent.xyz, 0.0)).xyz), in.tangent.w);
451
+ return out;
452
+ }
453
+
454
+ // Screen-space-derivative TBN. Reconstructs a tangent frame purely
455
+ // from the fragment's world-space position and UV — no vertex tangent
456
+ // attribute required. Based on Mikkelsen 2010 ('Followup: Normal
457
+ // Mapping Without Precomputed Tangents'). Gives close-to-identical
458
+ // results to pre-baked tangents for continuous UV mappings, which is
459
+ // the common case for PBR assets. We use this as a fallback when the
460
+ // mesh has no TANGENT accessor (very common — e.g., DamagedHelmet).
461
+ // The four screen-space derivatives are taken by the CALLER in uniform
462
+ // control flow and passed in: this function is reached from the per-fragment
463
+ // "mesh has no tangents" branch, and WGSL's uniformity analysis (enforced by
464
+ // Tint on WebGPU) rejects dpdx/dpdy inside non-uniform flow.
465
+ fn compute_tbn(dp1: vec3<f32>, dp2: vec3<f32>, duv1: vec2<f32>, duv2: vec2<f32>, n: vec3<f32>) -> mat3x3<f32> {
466
+ let dp2perp = cross(dp2, n);
467
+ let dp1perp = cross(n, dp1);
468
+ let t = dp2perp * duv1.x + dp1perp * duv2.x;
469
+ let b = dp2perp * duv1.y + dp1perp * duv2.y;
470
+ let denom = max(dot(t, t), dot(b, b));
471
+ let invmax = inverseSqrt(max(denom, 1e-20));
472
+ return mat3x3<f32>(t * invmax, b * invmax, n);
473
+ }
474
+
475
+ // Exact piecewise sRGB → linear, matching bloom-reference's
476
+ // `srgb_u8_to_linear`. The 2.2-gamma approximation we used before
477
+ // drifts by ~0.005 in mid-tones, which adds up across base color +
478
+ // emissive samples and skews IBL diffuse colors slightly bluer than
479
+ // the reference.
480
+ fn srgb_to_linear_v(c: vec3<f32>) -> vec3<f32> {
481
+ let cutoff = vec3<f32>(0.04045);
482
+ let lo = c / 12.92;
483
+ let hi = pow(max((c + vec3<f32>(0.055)) / 1.055, vec3<f32>(0.0)), vec3<f32>(2.4));
484
+ return select(hi, lo, c <= cutoff);
485
+ }
486
+
487
+ fn aces_tone(c: vec3<f32>) -> vec3<f32> {
488
+ let a = 2.51;
489
+ let b = 0.03;
490
+ let cc = 2.43;
491
+ let d = 0.59;
492
+ let e = 0.14;
493
+ return clamp((c * (c * a + b)) / (c * (c * cc + d) + e), vec3<f32>(0.0), vec3<f32>(1.0));
494
+ }
495
+
496
+ // --- Cook-Torrance GGX building blocks ---
497
+ fn d_ggx(n_dot_h: f32, alpha2: f32) -> f32 {
498
+ let x = n_dot_h * n_dot_h * (alpha2 - 1.0) + 1.0;
499
+ return alpha2 / (PI * x * x);
500
+ }
501
+
502
+ fn v_smith_ggx_correlated(n_dot_l: f32, n_dot_v: f32, alpha2: f32) -> f32 {
503
+ // Height-correlated Smith visibility (Heitz 2014). Combines with
504
+ // the Cook-Torrance /4*NdotL*NdotV denominator — so specular is
505
+ // D * V * F directly (no further divide).
506
+ let ggxv = n_dot_l * sqrt(n_dot_v * n_dot_v * (1.0 - alpha2) + alpha2);
507
+ let ggxl = n_dot_v * sqrt(n_dot_l * n_dot_l * (1.0 - alpha2) + alpha2);
508
+ return 0.5 / max(ggxv + ggxl, 1e-5);
509
+ }
510
+
511
+ fn f_schlick(v_dot_h: f32, f0: vec3<f32>) -> vec3<f32> {
512
+ let fc = pow(clamp(1.0 - v_dot_h, 0.0, 1.0), 5.0);
513
+ return f0 + (vec3<f32>(1.0) - f0) * fc;
514
+ }
515
+
516
+ // Sample a single cascade's shadow texture with 4-tap Poisson PCF.
517
+ fn sample_cascade(cascade: i32, shadow_uv: vec2<f32>, depth_ref: f32) -> f32 {
518
+ var dims: vec2<u32>;
519
+ if (cascade == 0) {
520
+ dims = textureDimensions(shadow_tex_0);
521
+ } else if (cascade == 1) {
522
+ dims = textureDimensions(shadow_tex_1);
523
+ } else {
524
+ dims = textureDimensions(shadow_tex_2);
525
+ }
526
+ let texel = vec2<f32>(1.0 / f32(dims.x), 1.0 / f32(dims.y));
527
+ // Tighter PCF radius (1.0 vs. prior 2.0). Softer was safer against
528
+ // shadow acne / swim but produced a ~4-texel penumbra on every
529
+ // shadow — for outdoor sun at this map resolution that translates
530
+ // to 2-3m of fuzz, which reads as 'painted' rather than 'cast'.
531
+ // The sun's real angular size gives a ~1m penumbra at typical
532
+ // scene distances; r=1.0 roughly matches that.
533
+ let radius = 1.0;
534
+ var sum = 0.0;
535
+ let poisson = array<vec2<f32>, 16>(
536
+ vec2<f32>(-0.94201624, -0.39906216),
537
+ vec2<f32>( 0.94558609, -0.76890725),
538
+ vec2<f32>(-0.09418410, -0.92938870),
539
+ vec2<f32>( 0.34495938, 0.29387760),
540
+ vec2<f32>(-0.91588581, 0.45771432),
541
+ vec2<f32>(-0.81544232, -0.87912464),
542
+ vec2<f32>(-0.38277543, 0.27676845),
543
+ vec2<f32>( 0.97484398, 0.75648379),
544
+ vec2<f32>( 0.44323325, -0.97511554),
545
+ vec2<f32>( 0.53742981, -0.47373420),
546
+ vec2<f32>(-0.26496911, -0.41893023),
547
+ vec2<f32>( 0.79197514, 0.19090188),
548
+ vec2<f32>(-0.24188840, 0.99706507),
549
+ vec2<f32>(-0.81409955, 0.91437590),
550
+ vec2<f32>( 0.19984126, 0.78641367),
551
+ vec2<f32>( 0.14383161, -0.14100790),
552
+ );
553
+ for (var i: i32 = 0; i < 16; i = i + 1) {
554
+ let off = poisson[i] * texel * radius;
555
+ let uv = shadow_uv + off;
556
+ if (cascade == 0) {
557
+ sum += textureSampleCompareLevel(shadow_tex_0, shadow_samp, uv, depth_ref);
558
+ } else if (cascade == 1) {
559
+ sum += textureSampleCompareLevel(shadow_tex_1, shadow_samp, uv, depth_ref);
560
+ } else {
561
+ sum += textureSampleCompareLevel(shadow_tex_2, shadow_samp, uv, depth_ref);
562
+ }
563
+ }
564
+ return sum / 16.0;
565
+ }
566
+
567
+ // Cascaded shadow map sampling. Determines which cascade the fragment
568
+ // belongs to based on its view-space depth, projects through that
569
+ // cascade's VP, and performs PCF. Blends between cascades at boundaries
570
+ // for smooth transitions.
571
+ fn sample_shadow(world_pos: vec3<f32>, geo_n: vec3<f32>) -> f32 {
572
+ // shadows disabled → fully lit. dir_light_count.y carries the enabled
573
+ // flag (splits.w is the TSR mip-LOD bias — do NOT gate on it); without
574
+ // this gate the projection below runs through identity/stale cascade
575
+ // VPs and the garbage NDC reads as 'occluded', so turning shadows OFF
576
+ // used to DARKEN ambient instead of removing shadows.
577
+ if (lighting.dir_light_count.y < 0.5) {
578
+ return 1.0;
579
+ }
580
+ // Select cascade by world-space DISTANCE from camera (not
581
+ // view-space Z). Distance is rotation-independent — spinning
582
+ // the camera doesn't change which cascade a surface falls in.
583
+ let cam = lighting.camera_pos.xyz;
584
+ let dist = length(world_pos - cam);
585
+
586
+ var cascade = 2;
587
+ if (dist <= lighting.shadow_cascade_splits.x) {
588
+ cascade = 0;
589
+ } else if (dist <= lighting.shadow_cascade_splits.y) {
590
+ cascade = 1;
591
+ }
592
+
593
+ // Normal-offset receiver bias: push the receiver position off the
594
+ // surface along its geometric normal by ~1.5 shadow texels of the
595
+ // selected cascade before projecting. The fixed depth bias alone
596
+ // (0.001 ≈ 8 cm across cascade 2's depth range) is SMALLER than the
597
+ // per-texel depth slope of steep receivers — a vertical wall under a
598
+ // 40°-elevation sun changes ~12 cm of light-space depth per shadow
599
+ // texel — so entire sun-facing faces used to self-shadow into a
600
+ // uniform ~50% PCF dimming (measured 68 vs 127 luma on the shooter's
601
+ // stone house). Offsetting the receiver sidesteps the slope entirely;
602
+ // the offset is texel-proportional (≈2 cm near, ≈23 cm at cascade 2),
603
+ // far below visible peter-panning at each cascade's viewing distance.
604
+ // The cascade fit radius ≈ its split distance (compute_cascade_vps
605
+ // fits a camera-centred sphere), so texel ≈ 2·split / map_dim.
606
+ let map_dim = f32(textureDimensions(shadow_tex_0).x);
607
+ var fit_r = lighting.shadow_cascade_splits.z;
608
+ if (cascade == 0) {
609
+ fit_r = lighting.shadow_cascade_splits.x;
610
+ } else if (cascade == 1) {
611
+ fit_r = lighting.shadow_cascade_splits.y;
612
+ }
613
+ let recv_pos = world_pos + geo_n * (2.0 * fit_r / map_dim) * 1.5;
614
+
615
+ // Project through the selected cascade's VP
616
+ let light_clip = lighting.shadow_cascade_vps[cascade] * vec4<f32>(recv_pos, 1.0);
617
+ let light_ndc = light_clip.xyz / light_clip.w;
618
+ if (light_ndc.x < -1.0 || light_ndc.x > 1.0 ||
619
+ light_ndc.y < -1.0 || light_ndc.y > 1.0 ||
620
+ light_ndc.z < 0.0 || light_ndc.z > 1.0) {
621
+ return 1.0;
622
+ }
623
+ let shadow_uv = vec2<f32>(light_ndc.x * 0.5 + 0.5, 1.0 - (light_ndc.y * 0.5 + 0.5));
624
+ let bias = 0.001;
625
+ let depth_ref = light_ndc.z - bias;
626
+ let shadow_val = sample_cascade(cascade, shadow_uv, depth_ref);
627
+
628
+ // Blend between cascades at boundary regions for smooth transitions.
629
+ // The blend zone is 10% of each cascade's range.
630
+ var split_near = 0.0;
631
+ var split_far = lighting.shadow_cascade_splits.x;
632
+ if (cascade == 1) {
633
+ split_near = lighting.shadow_cascade_splits.x;
634
+ split_far = lighting.shadow_cascade_splits.y;
635
+ } else if (cascade == 2) {
636
+ split_near = lighting.shadow_cascade_splits.y;
637
+ split_far = lighting.shadow_cascade_splits.z;
638
+ }
639
+ let blend_zone = (split_far - split_near) * 0.1;
640
+ let dist_to_edge = split_far - dist;
641
+
642
+ if (dist_to_edge < blend_zone && cascade < 2) {
643
+ // In the blend zone: sample the next cascade too and lerp.
644
+ // Same normal-offset receiver bias, scaled to the NEXT cascade's
645
+ // texel size (it is coarser, so the offset grows accordingly).
646
+ let next_cascade = cascade + 1;
647
+ var next_fit = lighting.shadow_cascade_splits.z;
648
+ if (next_cascade == 1) {
649
+ next_fit = lighting.shadow_cascade_splits.y;
650
+ }
651
+ let next_pos = world_pos + geo_n * (2.0 * next_fit / map_dim) * 1.5;
652
+ let next_clip = lighting.shadow_cascade_vps[next_cascade] * vec4<f32>(next_pos, 1.0);
653
+ let next_ndc = next_clip.xyz / next_clip.w;
654
+ let next_uv = vec2<f32>(next_ndc.x * 0.5 + 0.5, 1.0 - (next_ndc.y * 0.5 + 0.5));
655
+ let next_depth_ref = next_ndc.z - bias;
656
+ let next_val = sample_cascade(next_cascade, next_uv, next_depth_ref);
657
+ let t = dist_to_edge / blend_zone;
658
+ return mix(next_val, shadow_val, t);
659
+ }
660
+
661
+ return shadow_val;
662
+ }
663
+
664
+ // Evaluate a single directional light's PBR contribution. Returns
665
+ // linear-space radiance. `l_dir` points *from surface to light*,
666
+ // `intensity` scales the light color.
667
+ fn shade_pbr(
668
+ n: vec3<f32>,
669
+ v: vec3<f32>,
670
+ l_dir: vec3<f32>,
671
+ light_color: vec3<f32>,
672
+ intensity: f32,
673
+ base_color: vec3<f32>,
674
+ metallic: f32,
675
+ roughness: f32,
676
+ ) -> vec3<f32> {
677
+ let n_dot_l = max(dot(n, l_dir), 0.0);
678
+ if (n_dot_l <= 0.0 || intensity <= 0.0) {
679
+ return vec3<f32>(0.0);
680
+ }
681
+ let n_dot_v = max(dot(n, v), 1e-4);
682
+ // `normalize(0)` is NaN. At grazing-back angles (view roughly
683
+ // anti-parallel to the light direction on a near-flat surface)
684
+ // l + v can reach a vector indistinguishable from zero in f32,
685
+ // and a single NaN here survives the rest of the BRDF +
686
+ // tonemap chain as a pink speck. Skip the specular lobe when
687
+ // the half-vector is degenerate — diffuse still contributes.
688
+ let h_raw = l_dir + v;
689
+ let h_len2 = dot(h_raw, h_raw);
690
+ if (h_len2 <= 1e-12) {
691
+ let kd0 = (vec3<f32>(1.0) - mix(vec3<f32>(0.04), base_color, metallic)) * (1.0 - metallic);
692
+ return kd0 * base_color / PI * light_color * intensity * n_dot_l;
693
+ }
694
+ let h = h_raw * inverseSqrt(h_len2);
695
+ let n_dot_h = clamp(dot(n, h), 0.0, 1.0);
696
+ let v_dot_h = clamp(dot(v, h), 0.0, 1.0);
697
+
698
+ let alpha = max(roughness * roughness, 0.001);
699
+ let alpha2 = alpha * alpha;
700
+
701
+ let f0 = mix(vec3<f32>(0.04), base_color, metallic);
702
+ let f = f_schlick(v_dot_h, f0);
703
+ let d = d_ggx(n_dot_h, alpha2);
704
+ let vis = v_smith_ggx_correlated(n_dot_l, n_dot_v, alpha2);
705
+
706
+ let specular_raw = d * vis * f;
707
+
708
+ // Dielectric direct-specular attenuation. A polished marble column
709
+ // lit by the sun produces a narrow GGX highlight peak that pathtracers
710
+ // average over hemisphere-sized light sources; our point sun spikes
711
+ // D_GGX to 1000+ at the peak and survives tonemap as a bright stripe
712
+ // even after Fresnel (Intel Sponza column vs Cycles was the test).
713
+ // Same smoothstep-by-roughness treatment as the IBL path, applied
714
+ // only to the specular lobe — diffuse stays physically correct.
715
+ let dielectric_direct_amp = smoothstep(0.0, 1.0, roughness);
716
+ let dielectric_factor = 1.0 - metallic;
717
+ let direct_spec_scale = mix(1.0, dielectric_direct_amp, dielectric_factor);
718
+ // Universal roughness damping on direct specular too — same
719
+ // reasoning as the IBL path; a smooth marble column lit by a
720
+ // point sun spikes D_GGX past any tonemap cap. Metals stay at
721
+ // full direct spec for roughness > ~0.75.
722
+ // Universal soft luma cap on direct specular. A smooth marble
723
+ // cylinder hit by the sun spikes D_GGX past any reasonable
724
+ // tonemap; Reinhard-compress the luma toward a 0.3 ceiling
725
+ // smoothly so adjacent pixels with slightly different GGX peaks
726
+ // scale by neighbouring cap values instead of ping-ponging
727
+ // across a hard min() discontinuity (the cause of the sparkle on
728
+ // Sponza's sunlit floor tiles).
729
+ let direct_luma = dot(specular_raw, vec3<f32>(0.2126, 0.7152, 0.0722));
730
+ let direct_cap = 1.0 / (1.0 + direct_luma / 0.3);
731
+ let universal_damp = smoothstep(0.05, 0.75, roughness);
732
+ let specular = specular_raw * direct_spec_scale * universal_damp * direct_cap;
733
+
734
+ let kd = (vec3<f32>(1.0) - f) * (1.0 - metallic);
735
+ let diffuse = kd * base_color / PI;
736
+
737
+ return (diffuse + specular) * light_color * intensity * n_dot_l;
738
+ }
739
+
740
+ struct SceneOut {
741
+ @location(0) color: vec4<f32>,
742
+ @location(1) material: vec2<f32>,
743
+ @location(2) velocity: vec2<f32>,
744
+ /// Diffuse albedo (gamma-encoded base color). Used by post-passes
745
+ /// (SSGI, SSR) to modulate bounce light correctly — indirect
746
+ /// diffuse arriving at a surface is albedo × irradiance, not raw
747
+ /// radiance. Rgba8Unorm is enough precision here.
748
+ @location(3) albedo: vec4<f32>,
749
+ };
750
+
751
+ // EN-044 — depth prepass. Same vertex stage as the main pass (so the foliage wind
752
+ // displaces identically and the depths match), and a fragment stage that does
753
+ // nothing but honour the alpha cutout.
754
+ //
755
+ // WHY THIS EARNS ITS PASS. The scene fragment shader can `discard` (alpha-cutout
756
+ // foliage), and a shader that may discard cannot early-Z *write* — the GPU has to
757
+ // run the whole thing before it knows if the pixel survives. So every leaf card in
758
+ // an 88-tree forest shaded the full 5-target MRT, several layers deep, and threw
759
+ // most of it away. Priming depth first lets the main pass early-Z *reject* those
760
+ // fragments before the shader ever runs.
761
+ @fragment
762
+ fn fs_depth_prepass(in: VertexOutputScene) {
763
+ let alpha_cutoff = material.metal_rough.w;
764
+ if (alpha_cutoff > 0.0) {
765
+ let a = textureSample(base_color_tex, base_color_samp, in.uv).a * in.color.a;
766
+ if (a < alpha_cutoff) { discard; }
767
+ }
768
+ }
769
+
770
+ @fragment
771
+ fn fs_main_scene(in: VertexOutputScene) -> SceneOut {
772
+ var n = normalize(in.normal);
773
+
774
+ // --- Normal mapping (tangent-space) ---
775
+ // LEADR-lite normal map sample. The texture uploader bakes
776
+ // per-mip normal-direction variance into the alpha channel
777
+ // (see register_texture_kind). RGB holds the vector-averaged
778
+ // unit normal at each mip, so sampling any LOD gives a proper
779
+ // direction for shading; the alpha contains the accumulated
780
+ // (1 - |avg|²) disagreement across the footprint. The shader
781
+ // uses that alpha as an additional σ² term added to GGX α²,
782
+ // widening the lobe by exactly enough to integrate over sub-
783
+ // pixel normal variance before it hits the BRDF as sparkle.
784
+ //
785
+ // We still sample at +1 LOD bias so the hardware picks a mip
786
+ // with more accumulated variance than strictly minimal; the
787
+ // tradeoff is a hair of softness at near-perpendicular views
788
+ // in exchange for path-tracer-like integration at grazing.
789
+ // shadow_cascade_splits.w carries the global LOD bias (-1 when
790
+ // TSR is on, 0 otherwise) — added so half-res rendering still
791
+ // reads texture detail one mip finer than hardware would pick.
792
+ let lod_bias = lighting.shadow_cascade_splits.w;
793
+ let nm_sample4 = textureSampleBias(normal_tex, normal_samp, in.uv, 1.0 + lod_bias);
794
+ let nm_raw = nm_sample4.xyz * 2.0 - 1.0;
795
+ let baked_variance = nm_sample4.w;
796
+ let toksvig_len2 = clamp(dot(nm_raw, nm_raw), 0.01, 1.0);
797
+ let nm_sample = nm_raw * inverseSqrt(toksvig_len2);
798
+ // Derivatives for the no-tangent TBN fallback, taken here in uniform
799
+ // control flow (inside the branch below they would fail WGSL uniformity
800
+ // analysis on WebGPU).
801
+ let tbn_dp1 = dpdx(in.world_pos);
802
+ let tbn_dp2 = dpdy(in.world_pos);
803
+ let tbn_duv1 = dpdx(in.uv);
804
+ let tbn_duv2 = dpdy(in.uv);
805
+ let tlen2 = dot(in.tangent.xyz, in.tangent.xyz);
806
+ if (tlen2 > 0.0001) {
807
+ let t = normalize(in.tangent.xyz);
808
+ let t_ortho = normalize(t - n * dot(n, t));
809
+ let b = cross(n, t_ortho) * in.tangent.w;
810
+ n = normalize(t_ortho * nm_sample.x + b * nm_sample.y + n * nm_sample.z);
811
+ } else {
812
+ let tbn = compute_tbn(tbn_dp1, tbn_dp2, tbn_duv1, tbn_duv2, n);
813
+ n = normalize(tbn * nm_sample);
814
+ }
815
+
816
+ // --- Material sampling ---
817
+ // Base color & emissive textures in glTF are encoded as sRGB, but
818
+ // the bloom texture registrar creates them as Rgba8Unorm (no
819
+ // hardware decode). We decode manually via the 2.2 approximation —
820
+ // matches bloom-reference's convention so the PBR lighting math
821
+ // operates in linear space throughout.
822
+ let base_tex = textureSampleBias(base_color_tex, base_color_samp, in.uv, lod_bias);
823
+ // Vertex color carries the glTF baseColorFactor (linear per spec)
824
+ // when no per-vertex COLOR_0 stream exists, or the linear color
825
+ // attribute when it does. Do NOT srgb-decode it — that gave
826
+ // correct output only in the boundary case where baseColorFactor
827
+ // was (1,1,1,1), and silently darkened every legitimate tint
828
+ // (Bistro's spec-gloss diffuse factors land in the 0.5–0.9 range
829
+ // where the double-conversion is visibly off).
830
+ let base_color = srgb_to_linear_v(base_tex.rgb) * in.color.rgb;
831
+ let base_alpha = base_tex.a * in.color.a;
832
+
833
+ // glTF MASK / BLEND alpha mode — discard fragments below the
834
+ // authored cutoff so alpha-cutout foliage, fences, chains, and
835
+ // fabric render as their actual shape instead of opaque billboards.
836
+ // OPAQUE materials carry cutoff = 0 so the branch collapses.
837
+ // BLEND is treated as MASK @ 0.5 (via the loader) pending a real
838
+ // sorted transparent pipeline.
839
+ let alpha_cutoff = material.metal_rough.w;
840
+ if (alpha_cutoff > 0.0 && base_alpha < alpha_cutoff) {
841
+ discard;
842
+ }
843
+
844
+ // Two-sided foliage normal. Alpha-cutout cards (leaves, grass blades)
845
+ // are seen from both sides, but the geometric normal only faces one
846
+ // way — the back side otherwise shades with N pointing away from the
847
+ // sun AND from the sky irradiance, which is why grass tufts rendered
848
+ // as solid black cards from one side. Flip the shading normal toward
849
+ // the viewer for cutout materials only; opaque geometry is untouched.
850
+ if (alpha_cutoff > 0.0 && dot(n, lighting.camera_pos.xyz - in.world_pos) < 0.0) {
851
+ n = -n;
852
+ }
853
+
854
+ // glTF metallicRoughnessTexture: G=roughness, B=metallic (linear).
855
+ // When the material has no MR texture (metal_rough.z == 0), the
856
+ // binding falls back to an arbitrary scene texture (whatever lives
857
+ // at index 0) — multiplying its random R/G/B into our factors
858
+ // produces incorrect material values. Use the factors directly in
859
+ // that case.
860
+ let mr_tex_sample = textureSample(mr_tex, mr_samp, in.uv);
861
+ let has_mr = material.metal_rough.z > 0.5;
862
+ var roughness_raw = select(
863
+ clamp(material.metal_rough.y, 0.045, 1.0),
864
+ clamp(mr_tex_sample.g * material.metal_rough.y, 0.045, 1.0),
865
+ has_mr,
866
+ );
867
+ // Dielectric roughness floor. Real-world stone, wood, plaster etc.
868
+ // rarely get below ~0.15; when FBX2glTF or similar exporters drop
869
+ // them to 0.05, we get a mirror-like highlight strip on marble
870
+ // columns that Cycles doesn't produce (Sponza column was the tell).
871
+ // Metals keep the original low floor so chrome / gold stay sharp.
872
+ let metallic_raw = select(
873
+ clamp(material.metal_rough.x, 0.0, 1.0),
874
+ clamp(mr_tex_sample.b * material.metal_rough.x, 0.0, 1.0),
875
+ has_mr,
876
+ );
877
+ let metallic = metallic_raw;
878
+ let dielectric_floor = 0.15;
879
+ var roughness = max(roughness_raw,
880
+ dielectric_floor * (1.0 - metallic));
881
+
882
+ // Specular antialiasing. Two sources of variance are folded into
883
+ // GGX α² as additive corrections:
884
+ //
885
+ // 1. Toksvig (Kaplanyan 2016) — texture-level normal variance.
886
+ // The bilinearly-filtered+mipmapped normal map sample has
887
+ // length < 1 wherever adjacent normals disagree. σ² =
888
+ // (1 − r²)/r² is the Lambert-averaged normal variance,
889
+ // added directly to α² to widen the GGX lobe by exactly
890
+ // enough to integrate over the detail we can't resolve.
891
+ //
892
+ // 2. Screen-space kernel (Karis 2013) — geometry-level variance
893
+ // from per-pixel normal derivatives. Smaller cap than the
894
+ // pre-Toksvig version because Toksvig already handles the
895
+ // texture case; this term now only covers sharp geometric
896
+ // edges and tessellation that Toksvig can't see.
897
+ // Toksvig formula from the hardware-bilinear/aniso vector-length
898
+ // shortening, PLUS the per-mip variance baked into alpha during
899
+ // normal-map upload. The baked term is the clean directional-
900
+ // variance estimate; Toksvig adds whatever extra shortening the
901
+ // sampler's bilinear blend produces on top.
902
+ let sigma2_toksvig = (1.0 - toksvig_len2) / toksvig_len2;
903
+ let sigma2_baked = baked_variance / max(1.0 - baked_variance, 0.001);
904
+ let sigma2 = sigma2_toksvig + sigma2_baked;
905
+ var alpha2 = roughness * roughness + sigma2;
906
+ let nm_dx = dpdx(n);
907
+ let nm_dy = dpdy(n);
908
+ let curvature_sq = dot(nm_dx, nm_dx) + dot(nm_dy, nm_dy);
909
+ // Kaplanyan 2016 screen-space kernel. Bumped aggressively: 2.0
910
+ // coefficient / cap 0.9 to kill sparkle on Intel Sponza's sunlit
911
+ // floor tiles where each tile edge has a high-frequency normal-
912
+ // map bump that D_GGX spikes on at a grazing view. Integrates
913
+ // normal variance across a larger screen-space footprint before
914
+ // the BRDF sees it. Tradeoff: subtly softer micro-specular on
915
+ // all surfaces, which matches the path-tracer's multi-ray
916
+ // average.
917
+ let kernel_alpha = min(2.0 * curvature_sq, 0.9);
918
+ alpha2 = min(alpha2 + kernel_alpha, 1.0);
919
+ roughness = sqrt(alpha2);
920
+
921
+ let em_tex_sample = textureSample(em_tex, em_samp, in.uv);
922
+ let emissive = srgb_to_linear_v(em_tex_sample.rgb) * material.emissive.rgb;
923
+
924
+ // glTF occlusion: R channel, attenuates indirect lighting (IBL
925
+ // diffuse + ambient) only — direct lights and specular IBL are
926
+ // unchanged per spec. Default texture is white (idx 0) so the
927
+ // sample is 1.0 for materials without an occlusion map.
928
+ let occlusion = textureSample(occ_tex, occ_samp, in.uv).r;
929
+
930
+ // --- PBR direct lighting ---
931
+ let v = normalize(lighting.camera_pos.xyz - in.world_pos);
932
+ // Seed with ambient light contribution, modulated by base color
933
+ // so white walls pick up a white ambient and darker materials
934
+ // don't get over-brightened. This is the base illumination for
935
+ // surfaces that receive no direct light and are outside the IBL
936
+ // environment's strongest region (e.g. shadowed interiors).
937
+ var lit = lighting.ambient.rgb * lighting.ambient.a * base_color;
938
+
939
+ // Legacy primary directional (kept for back-compat). Shadow-
940
+ // mapped: only this primary light casts because we currently
941
+ // render a single shadow map. Multi-cascade or multi-light
942
+ // shadowing is a future addition.
943
+ // Geometric (pre-normal-map, pre-foliage-flip) normal for the receiver
944
+ // offset — the mapped normal can point anywhere per-texel and would
945
+ // dither the offset; the flipped foliage normal would push the sample
946
+ // through the card.
947
+ let shadow_factor = sample_shadow(in.world_pos, normalize(in.normal));
948
+ // Never fully zero direct light — a 10% floor simulates
949
+ // ambient bounce from surrounding surfaces and keeps shadows
950
+ // from going pitch-black regardless of IBL intensity.
951
+ let direct_shadow_raw = mix(0.03, 1.0, shadow_factor);
952
+ let legacy_dir = normalize(lighting.light_dir.xyz);
953
+ // Cloud deck (common/clouds.wgsl). Folded into the SUN shadow only: a cloud
954
+ // blocks the sun, it does not stop the sky from being blue. Multiplying it
955
+ // into ambient as well is what makes cloud shadows read as flat grey paint
956
+ // instead of shade. Costs nothing when strength is 0 (the default).
957
+ let direct_shadow = direct_shadow_raw * cloud_shadow_at(
958
+ in.world_pos, legacy_dir, lighting.wind.xy, lighting.wind.w, lighting.cloud);
959
+ if (alpha_cutoff > 0.0) {
960
+ // Foliage wrap-lambert (energy-conserving wrap, w = 0.45): a leaf
961
+ // turning from the sun rolls off softly — light transmits and
962
+ // inter-scatters through a canopy — instead of clipping to black
963
+ // at the terminator like an opaque wall. Specular is skipped:
964
+ // foliage cards are rough and the viewer-flipped normal would
965
+ // produce false sparkle.
966
+ let wrap = 0.45;
967
+ let ndl_wrap = clamp((dot(n, legacy_dir) + wrap) / ((1.0 + wrap) * (1.0 + wrap)),
968
+ 0.0, 1.0);
969
+ lit += base_color / PI * lighting.light_color.rgb * lighting.light_dir.w
970
+ * ndl_wrap * direct_shadow;
971
+ } else {
972
+ lit += shade_pbr(n, v, legacy_dir, lighting.light_color.rgb,
973
+ lighting.light_dir.w, base_color, metallic, roughness)
974
+ * direct_shadow;
975
+ }
976
+
977
+ // Foliage backlit transmission — sun bleeding THROUGH alpha-cut leaf cards
978
+ // (the bright rim glow when the sun is behind a tree). Gated on the
979
+ // alpha-cutoff so only cut-out foliage materials get it; opaque surfaces
980
+ // (cutoff == 0) are unaffected. Matches shade_foliage's transmission term.
981
+ // Round-2 audit: this block was pasted TWICE (1.7x strength) and ran
982
+ // unshadowed — a canopy in another tree's shadow still glowed at full
983
+ // transmission. De-duplicated and multiplied by the sun shadow factor.
984
+ if (alpha_cutoff > 0.0) {
985
+ let trans = pow(max(dot(v, -legacy_dir), 0.0), 3.0) * 0.85;
986
+ lit += base_color * lighting.light_color.rgb * lighting.light_dir.w * trans
987
+ * direct_shadow;
988
+ }
989
+
990
+ let dir_count = u32(lighting.dir_light_count.x);
991
+ for (var i = 0u; i < dir_count; i++) {
992
+ let dl = lighting.dir_lights[i];
993
+ let l = normalize(dl.direction.xyz);
994
+ lit += shade_pbr(n, v, l, dl.color.rgb, dl.direction.w,
995
+ base_color, metallic, roughness);
996
+ }
997
+
998
+ // BEGIN-POINT-LIGHT-LOOP (replaced by the froxel-clustered variant
999
+ // at pipeline build on storage-buffer-capable backends — see
1000
+ // renderer/froxel.rs; this plain loop is the WebGL fallback and the
1001
+ // semantic reference the clustered path must match exactly)
1002
+ let pt_count = u32(lighting.point_light_count.x);
1003
+ for (var i = 0u; i < pt_count; i++) {
1004
+ let pl = lighting.point_lights[i];
1005
+ let to_light = pl.position.xyz - in.world_pos;
1006
+ let dist = length(to_light);
1007
+ let range = pl.position.w;
1008
+ if (dist < range && dist > 0.0) {
1009
+ let l = to_light / dist;
1010
+ let atten = 1.0 - (dist / range);
1011
+ let atten2 = atten * atten;
1012
+ lit += shade_pbr(n, v, l, pl.color.rgb, pl.color.w * atten2,
1013
+ base_color, metallic, roughness);
1014
+ }
1015
+ }
1016
+ // END-POINT-LIGHT-LOOP
1017
+
1018
+ // --- Split-sum IBL (Karis 2013) ---
1019
+ // IBL_diffuse = base_color * (1 - kS_avg) * (1 - metallic)
1020
+ // * env_irradiance(N)
1021
+ // IBL_specular = prefiltered_env(R, roughness)
1022
+ // * (F0 * brdf.scale + brdf.bias)
1023
+ //
1024
+ // env_irradiance is approximated by sampling the env map at its
1025
+ // smallest mip (heaviest blur — close enough to a cosine-
1026
+ // convolved irradiance map for low-frequency diffuse lighting).
1027
+ // prefiltered_env samples mip = roughness * (mips-1), where the
1028
+ // mip chain was box-filter downsampled. Box filter ≠ true GGX
1029
+ // convolution — that's the next refinement — but together with
1030
+ // the BRDF LUT it captures the bulk of correct PBR appearance.
1031
+
1032
+ let n_dot_v_ibl = max(dot(n, v), 0.0);
1033
+ let f0 = mix(vec3<f32>(0.04), base_color, metallic);
1034
+
1035
+ // Diffuse irradiance: dedicated cosine-convolved texture populated
1036
+ // at env load. Sampling it directly (mip 0) at the fragment normal
1037
+ // gives proper Lambertian diffuse — no mip-steal hack on the
1038
+ // specular chain, so specular can use every mip for GGX prefilter.
1039
+ let mips = f32(textureNumLevels(env_tex));
1040
+ let irr_uv = seamless_equirect_uv(dir_to_equirect_uv(n));
1041
+ let irradiance = textureSampleLevel(env_diffuse_tex, env_samp, irr_uv, 0.0).rgb
1042
+ * lighting.camera_pos.w;
1043
+
1044
+ // For diffuse IBL, the Schlick-with-roughness approximation
1045
+ // (Lazarov 2013) handles the average kS factor at grazing angles.
1046
+ let fc_n = pow(1.0 - n_dot_v_ibl, 5.0);
1047
+ let f_ibl = f0 + (max(vec3<f32>(1.0 - roughness), f0) - f0) * fc_n;
1048
+ let kd = (vec3<f32>(1.0) - f_ibl) * (1.0 - metallic);
1049
+ let ibl_diffuse = irradiance * base_color * kd * occlusion;
1050
+
1051
+ // Pre-filtered specular sample at mip = roughness * (mips - 1).
1052
+ // All env_tex mips are GGX-prefiltered now that diffuse lives in
1053
+ // its own dedicated texture — roughness = 1 samples the smallest,
1054
+ // most-blurred mip, and roughness = 0 samples mip 0 (mirror).
1055
+ let r = reflect(-v, n);
1056
+ let max_spec_mip = max(mips - 1.0, 0.0);
1057
+ let prefiltered_env = env_sample_lod(r, roughness * max_spec_mip);
1058
+
1059
+ // BRDF LUT lookup — (NdotV, roughness) → (scale, bias) such that
1060
+ // single-scatter specular = env * (F0 * scale + bias).
1061
+ // Pre-integrated against GGX so the directional integral is correct.
1062
+ let brdf = textureSample(brdf_lut_tex, brdf_lut_samp, vec2<f32>(n_dot_v_ibl, roughness)).rg;
1063
+ let single_spec = prefiltered_env * (f0 * brdf.x + vec3<f32>(brdf.y));
1064
+
1065
+ // Multi-scattering compensation (Fdez-Aguera 2019). Single-scatter
1066
+ // GGX loses energy at high roughness — light that should bounce
1067
+ // around the microsurface gets dropped. We add it back as a second
1068
+ // term tinted by F0 * average-scatter, using the BRDF LUT energy
1069
+ // total (brdf.x + brdf.y) as 'how much energy did single-scatter
1070
+ // capture' so 1 - that_total is what we missed. Visually: rough
1071
+ // metals (gold, copper) get noticeably brighter and more saturated.
1072
+ // Multi-scatter compensation (Fdez-Aguera 2019, proper form).
1073
+ // E_ss = brdf.x + brdf.y single-scatter energy
1074
+ // E_ms = 1 - E_ss missing (multi-scatter) energy
1075
+ // F_avg = F0 + (1-F0)/21 average fresnel (Karis)
1076
+ // F_ms = F_avg * E_ss / (1 - F_avg * E_ms) multi-scatter fresnel
1077
+ // ms = F_ms * E_ms extra radiance to add back
1078
+ // The previous simpler form `1 + f_avg*(1/E_ss - 1)` exploded
1079
+ // as E_ss → 0 (rough dielectrics at grazing), blowing the
1080
+ // ground out to white.
1081
+ let ess = brdf.x + brdf.y;
1082
+ let ems = 1.0 - ess;
1083
+ let f_avg = f0 + (vec3<f32>(1.0) - f0) * (1.0 / 21.0);
1084
+ let f_ms = f_avg * ess / (vec3<f32>(1.0) - f_avg * ems);
1085
+ let ms_contribution = f_ms * ems;
1086
+
1087
+ // Specular occlusion (Lagarde 2014, Moving Frostbite to PBR):
1088
+ // attenuate IBL specular by a roughness-weighted blend of the glTF
1089
+ // AO term and NdotV so smooth dielectrics in enclosed/shadowed
1090
+ // cavities stop reflecting bright sky patches that no path-tracer
1091
+ // would let through the occluders. For metals and mirrors this is
1092
+ // near-identity; for rough surfaces it approaches the AO value.
1093
+ let spec_occ = clamp(
1094
+ pow(n_dot_v_ibl + occlusion, exp2(-16.0 * roughness - 1.0))
1095
+ - 1.0 + occlusion,
1096
+ 0.0, 1.0,
1097
+ );
1098
+ let ibl_spec_raw = prefiltered_env
1099
+ * (f0 * brdf.x + vec3<f32>(brdf.y) + ms_contribution);
1100
+
1101
+ // Dielectric specular luma cap. Without a proper visibility-aware
1102
+ // specular integral, smooth non-metals like marble / varnished wood
1103
+ // end up reflecting the HDR's bright-sky region at full intensity
1104
+ // even when occluded by intervening geometry (the Intel Sponza
1105
+ // column stripe vs Cycles was the smoking gun — proven by a
1106
+ // roughness=1 test render where the stripe disappeared).
1107
+ // Path-tracers handle this via shadow rays; we approximate by
1108
+ // (1) hard luma cap at 0.8 mid-grey — barely visible — and
1109
+ // (2) scaling the dielectric spec amplitude by roughness so the
1110
+ // polished end of the scale (roughness 0.15-0.35) loses almost all
1111
+ // of its IBL specular response. Metals are left alone so chrome
1112
+ // and gold keep their full dynamic range.
1113
+ let spec_luma = dot(ibl_spec_raw, vec3<f32>(0.2126, 0.7152, 0.0722));
1114
+ let dielectric_factor = 1.0 - metallic;
1115
+ let luma_cap = 0.5;
1116
+ let cap_scale = select(1.0, luma_cap / max(spec_luma, 0.0001),
1117
+ spec_luma > luma_cap);
1118
+ // Roughness attenuation curve for dielectrics: fully off on
1119
+ // polished surfaces, on-ramp all the way to roughness 1.0 where
1120
+ // the prefiltered blur covers a full hemisphere so a wrong sample
1121
+ // is guaranteed to average with its occluded neighbours. This
1122
+ // nearly wipes the column stripe without killing specular on
1123
+ // rough natural stone — matter with roughness 0.7 still gets
1124
+ // ~50% of the IBL spec contribution.
1125
+ let dielectric_spec_amp = smoothstep(0.0, 1.0, roughness);
1126
+ let dielectric_scale = mix(1.0, cap_scale * dielectric_spec_amp,
1127
+ dielectric_factor);
1128
+ // Universal roughness-based spec attenuation. A smooth curved
1129
+ // surface with ANY material (even metal) needs visibility-aware
1130
+ // specular to avoid bright stripes where the reflection vector
1131
+ // happens to sweep across a hot HDR sample. We don't have
1132
+ // visibility, so we dial down specular for smooth surfaces
1133
+ // regardless of metalness. Roughness 0.15 floor (applied upstream
1134
+ // to dielectrics) plus this smoothstep leaves roughness 1.0
1135
+ // surfaces untouched and mid-rough surfaces (0.3-0.5) at
1136
+ // significant but reduced strength. Metals still get the
1137
+ // dielectric_scale (via the metallic-weighted mix), so the
1138
+ // combined effect is conservative for both.
1139
+ // Universal luma cap: whatever the metallicity or the roughness,
1140
+ // the IBL specular contribution for a single fragment can't exceed
1141
+ // a hard luma ceiling. Marble / stone columns get their mirror-
1142
+ // of-a-bright-sky-strip reflection clipped to something that
1143
+ // couldn't survive a path-tracer's visibility integral,
1144
+ // and brightly-polished metals lose a little punch (they compensate
1145
+ // via direct specular which still uses Fresnel at full strength).
1146
+ // Reinhard-style soft luma cap: same 0.3 ceiling as direct spec
1147
+ // but smooth rolloff so adjacent pixels don't ping-pong across a
1148
+ // hard discontinuity (speckle on sunlit floor tiles with
1149
+ // per-pixel roughness / normal-map variation).
1150
+ let cap2_luma = dot(ibl_spec_raw, vec3<f32>(0.2126, 0.7152, 0.0722));
1151
+ let cap2 = 1.0 / (1.0 + cap2_luma / 0.3);
1152
+ let roughness_amp = smoothstep(0.05, 0.75, roughness);
1153
+ // EN-021 exclusive ownership: where SSR is active it owns specular —
1154
+ // hit (traced colour) or miss (env fallback inside the SSR shader).
1155
+ // Scale IBL specular by the complement of SSR's own roughness fade
1156
+ // × its strength (dir_light_count.z, written per frame; 0 when SSR
1157
+ // is disabled so the full IBL term returns). Kills the metal
1158
+ // double-count on hits (round-2 audit F10) without darkening
1159
+ // off-screen reflections.
1160
+ let ssr_own = clamp(
1161
+ lighting.dir_light_count.z * (1.0 - smoothstep(0.5, 0.85, roughness)),
1162
+ 0.0, 1.0);
1163
+ let ibl_spec = ibl_spec_raw
1164
+ * dielectric_scale * spec_occ * roughness_amp * cap2 * (1.0 - ssr_own);
1165
+
1166
+ // Indirect-shadow attenuation. 0.15 — deep enough that windows
1167
+ // Shadow darkening floor. Prior 0.15 matched Cycles path-
1168
+ // tracer output — physically correct, but visually heavy on
1169
+ // screens calibrated against UE5 / Unity renders, which
1170
+ // preserve more sky-bounce in shaded regions. 0.35 keeps
1171
+ // shadowed areas legible (Sponza atrium under-awning stays
1172
+ // 35 % of its indirect-light budget instead of 15 %) without
1173
+ // washing out the shadow line. Matches the general look of
1174
+ // UE5's Lumen + sky-occlusion and Unity HDRP's ambient
1175
+ // probes in Sponza/Bistro test scenes.
1176
+ let indirect_shadow = mix(0.35, 1.0, shadow_factor);
1177
+
1178
+ // Multi-scatter also adds a diffuse-like term back from the
1179
+ // 'lost' energy, but it gets absorbed wherever there is no metal
1180
+ // since dielectrics already account for it via the (1 - kS)
1181
+ // diffuse term. The compensation above handles the metal case;
1182
+ // dielectric path is unchanged.
1183
+ let hdr_raw = lit + (ibl_diffuse + ibl_spec) * indirect_shadow + emissive;
1184
+
1185
+ // Final HDR scrub. Two things the rest of the chain can't
1186
+ // recover from:
1187
+ //
1188
+ // 1. NaN/Inf anywhere upstream (unguarded GGX at α→0 +
1189
+ // n_dot_h→1, multi-scatter `1 / (1 - F_avg·E_ms)` at
1190
+ // grazing smooth metals, env-sample weirdness at UV seams)
1191
+ // — a single poisoned pixel survives TAA's neighborhood
1192
+ // clamp on Metal (clamp(NaN,a,b) is impl-defined) and
1193
+ // tonemaps to pink. Self-compare kills it at source.
1194
+ //
1195
+ // 2. Specular fireflies from sub-pixel normal-map variance.
1196
+ // The LEADR baked σ² already widens the GGX lobe by the
1197
+ // accumulated mip footprint, but there are still isolated
1198
+ // texels where D_GGX + IBL prefilter spike an order of
1199
+ // magnitude above neighbours. Bloom then amplifies each
1200
+ // spike into a coloured halo. The real root cause of the
1201
+ // stone-floor speckle was the irradiance convolution
1202
+ // shader sampling raw HDR (sun disc unclamped) — with
1203
+ // that fixed this cap only has to catch legitimate
1204
+ // specular outliers. 50 leaves all normal bright content
1205
+ // alone and trims only the rare aliased peak.
1206
+ let hdr_clean = select(vec3<f32>(0.0), hdr_raw, hdr_raw == hdr_raw);
1207
+ let luma = dot(hdr_clean, vec3<f32>(0.2126, 0.7152, 0.0722));
1208
+ let firefly_cap = 50.0;
1209
+ let luma_scale = select(1.0, firefly_cap / luma, luma > firefly_cap);
1210
+ let hdr = hdr_clean * luma_scale;
1211
+
1212
+ // Per-pixel velocity: difference between current and previous NDC,
1213
+ // scaled by 0.5 so the result is in UV-space units. Used by the
1214
+ // motion blur pass and TAA per-object reprojection.
1215
+ let curr_ndc = in.curr_clip.xy / in.curr_clip.w;
1216
+ let prev_ndc = in.prev_clip.xy / in.prev_clip.w;
1217
+ let vel = (curr_ndc - prev_ndc) * 0.5;
1218
+
1219
+ // glTF OPAQUE materials (alpha_cutoff == 0) ignore texture alpha by
1220
+ // spec — armor/gloss masks stored in .a must not make the mesh
1221
+ // translucent. MASK materials keep the sampled alpha (post-discard)
1222
+ // for soft edges. Tint alpha survives so games can fade models.
1223
+ let out_alpha = select(in.color.a, base_alpha, alpha_cutoff > 0.0);
1224
+
1225
+ return SceneOut(
1226
+ vec4<f32>(hdr, out_alpha),
1227
+ vec2<f32>(metallic, roughness),
1228
+ vel,
1229
+ // albedo.rgb: base color (SSGI bounce modulation).
1230
+ // albedo.a: 1 - shadow_factor — how much of this pixel's
1231
+ // illumination is INDIRECT (IBL + bounce) vs
1232
+ // DIRECT (sun). The compose pass uses this to
1233
+ // apply SSAO only to indirect-dominated pixels
1234
+ // (shadowed corners, overhangs) and leave
1235
+ // sun-lit surfaces alone, which is the physically
1236
+ // correct behaviour for AO (occludes indirect
1237
+ // only). 1.0 where fully shadowed, 0.0 where
1238
+ // sunlit. Sky shader overrides with 0.0.
1239
+ vec4<f32>(base_color, 1.0 - shadow_factor),
1240
+ );
1241
+ }
1242
+ "#);
1243
+