@bornengine/engine 0.4.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (213) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +231 -0
  3. package/native/android/Cargo.lock +1848 -0
  4. package/native/android/Cargo.toml +24 -0
  5. package/native/android/src/lib.rs +702 -0
  6. package/native/ios/Cargo.lock +1690 -0
  7. package/native/ios/Cargo.toml +32 -0
  8. package/native/ios/src/lib.rs +1267 -0
  9. package/native/linux/Cargo.lock +3279 -0
  10. package/native/linux/Cargo.toml +29 -0
  11. package/native/linux/src/lib.rs +1331 -0
  12. package/native/macos/Cargo.lock +3310 -0
  13. package/native/macos/Cargo.toml +46 -0
  14. package/native/macos/src/lib.rs +1302 -0
  15. package/native/shared/Cargo.lock +1899 -0
  16. package/native/shared/Cargo.toml +62 -0
  17. package/native/shared/assets/default_font.ttf +0 -0
  18. package/native/shared/build.rs +270 -0
  19. package/native/shared/shaders/common/clouds.wgsl +122 -0
  20. package/native/shared/shaders/common/fog.wgsl +16 -0
  21. package/native/shared/shaders/common/foliage_wind.wgsl +98 -0
  22. package/native/shared/shaders/common/imposter.wgsl +112 -0
  23. package/native/shared/shaders/common/pbr.wgsl +186 -0
  24. package/native/shared/shaders/common/shadows.wgsl +186 -0
  25. package/native/shared/shaders/common/sky.wgsl +8 -0
  26. package/native/shared/shaders/common/tonemap.wgsl +25 -0
  27. package/native/shared/shaders/impulse_field.wgsl +57 -0
  28. package/native/shared/shaders/material_abi.wgsl +383 -0
  29. package/native/shared/shaders/materials/test_minimal.wgsl +42 -0
  30. package/native/shared/src/anim_mixer.rs +61 -0
  31. package/native/shared/src/attach.rs +263 -0
  32. package/native/shared/src/audio/decode.rs +123 -0
  33. package/native/shared/src/audio/mod.rs +863 -0
  34. package/native/shared/src/audio/render.rs +892 -0
  35. package/native/shared/src/audio/spsc.rs +156 -0
  36. package/native/shared/src/audio/stream.rs +226 -0
  37. package/native/shared/src/custom_shaders.rs +104 -0
  38. package/native/shared/src/decals.rs +245 -0
  39. package/native/shared/src/drs.rs +211 -0
  40. package/native/shared/src/engine.rs +261 -0
  41. package/native/shared/src/ffi.rs +116 -0
  42. package/native/shared/src/ffi_core/assets.rs +388 -0
  43. package/native/shared/src/ffi_core/audio_ffi.rs +184 -0
  44. package/native/shared/src/ffi_core/draw.rs +334 -0
  45. package/native/shared/src/ffi_core/game_loop.rs +577 -0
  46. package/native/shared/src/ffi_core/input.rs +234 -0
  47. package/native/shared/src/ffi_core/mod.rs +127 -0
  48. package/native/shared/src/ffi_core/models.rs +1154 -0
  49. package/native/shared/src/ffi_core/ragdoll_ffi.rs +261 -0
  50. package/native/shared/src/ffi_core/scene.rs +626 -0
  51. package/native/shared/src/ffi_core/vfx.rs +212 -0
  52. package/native/shared/src/ffi_core/visual.rs +691 -0
  53. package/native/shared/src/frame_callbacks.rs +122 -0
  54. package/native/shared/src/geometry.rs +236 -0
  55. package/native/shared/src/handles.rs +182 -0
  56. package/native/shared/src/input.rs +448 -0
  57. package/native/shared/src/jolt_sys.rs +822 -0
  58. package/native/shared/src/lib.rs +55 -0
  59. package/native/shared/src/models.rs +1093 -0
  60. package/native/shared/src/models_gltf.rs +1280 -0
  61. package/native/shared/src/particles.rs +391 -0
  62. package/native/shared/src/physics_jolt.rs +1908 -0
  63. package/native/shared/src/picking.rs +298 -0
  64. package/native/shared/src/postfx.rs +345 -0
  65. package/native/shared/src/profiler.rs +492 -0
  66. package/native/shared/src/ragdoll.rs +474 -0
  67. package/native/shared/src/renderer/atmosphere_lut.rs +573 -0
  68. package/native/shared/src/renderer/brdf_lut.rs +154 -0
  69. package/native/shared/src/renderer/draw2d.rs +143 -0
  70. package/native/shared/src/renderer/formats.rs +822 -0
  71. package/native/shared/src/renderer/froxel.rs +421 -0
  72. package/native/shared/src/renderer/gi_bake.rs +653 -0
  73. package/native/shared/src/renderer/graph.rs +462 -0
  74. package/native/shared/src/renderer/hiz.rs +269 -0
  75. package/native/shared/src/renderer/hot_reload.rs +390 -0
  76. package/native/shared/src/renderer/impulse_field.rs +456 -0
  77. package/native/shared/src/renderer/lighting.rs +154 -0
  78. package/native/shared/src/renderer/material_instancing.rs +171 -0
  79. package/native/shared/src/renderer/material_pipeline.rs +700 -0
  80. package/native/shared/src/renderer/material_system.rs +1996 -0
  81. package/native/shared/src/renderer/material_system_tests.rs +601 -0
  82. package/native/shared/src/renderer/material_system_wasm.rs +41 -0
  83. package/native/shared/src/renderer/mod.rs +12556 -0
  84. package/native/shared/src/renderer/model_draw.rs +641 -0
  85. package/native/shared/src/renderer/occlusion.rs +429 -0
  86. package/native/shared/src/renderer/planar_pass.rs +593 -0
  87. package/native/shared/src/renderer/planar_reflection.rs +499 -0
  88. package/native/shared/src/renderer/post_pass.rs +249 -0
  89. package/native/shared/src/renderer/postfx_chain.rs +728 -0
  90. package/native/shared/src/renderer/pt_pass.rs +577 -0
  91. package/native/shared/src/renderer/scene_pass.rs +607 -0
  92. package/native/shared/src/renderer/shader_include.rs +205 -0
  93. package/native/shared/src/renderer/shader_library.rs +135 -0
  94. package/native/shared/src/renderer/shaders/ao.rs +570 -0
  95. package/native/shared/src/renderer/shaders/core.rs +1243 -0
  96. package/native/shared/src/renderer/shaders/env.rs +907 -0
  97. package/native/shared/src/renderer/shaders/gi.rs +810 -0
  98. package/native/shared/src/renderer/shaders/mod.rs +19 -0
  99. package/native/shared/src/renderer/shaders/post.rs +1558 -0
  100. package/native/shared/src/renderer/shaders/pt.rs +1859 -0
  101. package/native/shared/src/renderer/shaders/ssgi.rs +1586 -0
  102. package/native/shared/src/renderer/shadow_pass.rs +731 -0
  103. package/native/shared/src/renderer/ssgi_pass.rs +392 -0
  104. package/native/shared/src/renderer/ssr_pass.rs +188 -0
  105. package/native/shared/src/renderer/texture_store.rs +473 -0
  106. package/native/shared/src/renderer/transient.rs +591 -0
  107. package/native/shared/src/renderer/types.rs +941 -0
  108. package/native/shared/src/renderer/util.rs +152 -0
  109. package/native/shared/src/scene.rs +1362 -0
  110. package/native/shared/src/sdf_cache.rs +274 -0
  111. package/native/shared/src/shadows.rs +1036 -0
  112. package/native/shared/src/staging.rs +102 -0
  113. package/native/shared/src/string_header.rs +266 -0
  114. package/native/shared/src/text_renderer.rs +502 -0
  115. package/native/shared/src/textures.rs +197 -0
  116. package/native/tvos/Cargo.lock +1693 -0
  117. package/native/tvos/Cargo.toml +36 -0
  118. package/native/tvos/metal-patched/Cargo.toml +178 -0
  119. package/native/tvos/metal-patched/LICENSE-APACHE +201 -0
  120. package/native/tvos/metal-patched/LICENSE-MIT +25 -0
  121. package/native/tvos/metal-patched/src/acceleration_structure.rs +667 -0
  122. package/native/tvos/metal-patched/src/acceleration_structure_pass.rs +108 -0
  123. package/native/tvos/metal-patched/src/argument.rs +366 -0
  124. package/native/tvos/metal-patched/src/blitpass.rs +102 -0
  125. package/native/tvos/metal-patched/src/buffer.rs +71 -0
  126. package/native/tvos/metal-patched/src/capturedescriptor.rs +76 -0
  127. package/native/tvos/metal-patched/src/capturemanager.rs +113 -0
  128. package/native/tvos/metal-patched/src/commandbuffer.rs +192 -0
  129. package/native/tvos/metal-patched/src/commandqueue.rs +44 -0
  130. package/native/tvos/metal-patched/src/computepass.rs +107 -0
  131. package/native/tvos/metal-patched/src/constants.rs +152 -0
  132. package/native/tvos/metal-patched/src/counters.rs +119 -0
  133. package/native/tvos/metal-patched/src/depthstencil.rs +190 -0
  134. package/native/tvos/metal-patched/src/device.rs +2134 -0
  135. package/native/tvos/metal-patched/src/drawable.rs +39 -0
  136. package/native/tvos/metal-patched/src/encoder.rs +2041 -0
  137. package/native/tvos/metal-patched/src/heap.rs +281 -0
  138. package/native/tvos/metal-patched/src/indirect_encoder.rs +344 -0
  139. package/native/tvos/metal-patched/src/lib.rs +657 -0
  140. package/native/tvos/metal-patched/src/library.rs +902 -0
  141. package/native/tvos/metal-patched/src/mps.rs +575 -0
  142. package/native/tvos/metal-patched/src/pipeline/compute.rs +475 -0
  143. package/native/tvos/metal-patched/src/pipeline/mod.rs +71 -0
  144. package/native/tvos/metal-patched/src/pipeline/render.rs +762 -0
  145. package/native/tvos/metal-patched/src/renderpass.rs +443 -0
  146. package/native/tvos/metal-patched/src/resource.rs +182 -0
  147. package/native/tvos/metal-patched/src/sampler.rs +165 -0
  148. package/native/tvos/metal-patched/src/sync.rs +178 -0
  149. package/native/tvos/metal-patched/src/texture.rs +352 -0
  150. package/native/tvos/metal-patched/src/types.rs +90 -0
  151. package/native/tvos/metal-patched/src/vertexdescriptor.rs +250 -0
  152. package/native/tvos/src/audio_backend.rs +197 -0
  153. package/native/tvos/src/lib.rs +1891 -0
  154. package/native/visionos/Cargo.lock +1693 -0
  155. package/native/visionos/Cargo.toml +40 -0
  156. package/native/visionos/src/audio_backend.rs +197 -0
  157. package/native/visionos/src/lib.rs +1887 -0
  158. package/native/watchos/Cargo.lock +16 -0
  159. package/native/watchos/Cargo.toml +19 -0
  160. package/native/watchos/shaders/bloom_postfx.metal +99 -0
  161. package/native/watchos/src/BloomWatchApp.swift +1267 -0
  162. package/native/watchos/src/BloomWatchAudio.swift +179 -0
  163. package/native/watchos/src/audio.rs +55 -0
  164. package/native/watchos/src/draw_list.rs +229 -0
  165. package/native/watchos/src/ffi_stubs.rs +915 -0
  166. package/native/watchos/src/ffi_stubs_manual.rs +35 -0
  167. package/native/watchos/src/lib.rs +1124 -0
  168. package/native/watchos/src/models.rs +746 -0
  169. package/native/watchos/src/postfx.rs +95 -0
  170. package/native/watchos/src/scene.rs +534 -0
  171. package/native/watchos/src/textures.rs +184 -0
  172. package/native/web/Cargo.lock +1657 -0
  173. package/native/web/Cargo.toml +43 -0
  174. package/native/web/bloom_glue.js +695 -0
  175. package/native/web/build.sh +131 -0
  176. package/native/web/index.html +35 -0
  177. package/native/web/jolt_bridge.js +1519 -0
  178. package/native/web/src/input_ffi.rs +286 -0
  179. package/native/web/src/lib.rs +1796 -0
  180. package/native/web/src/material_ffi.rs +710 -0
  181. package/native/web/src/parity_ffi.rs +343 -0
  182. package/native/web/src/physics_ffi.rs +643 -0
  183. package/native/web/src/ragdoll_ffi.rs +250 -0
  184. package/native/web/src/render_settings.rs +98 -0
  185. package/native/windows/Cargo.lock +1815 -0
  186. package/native/windows/Cargo.toml +68 -0
  187. package/native/windows/src/lib.rs +1486 -0
  188. package/package.json +4279 -0
  189. package/src/audio/index.ts +315 -0
  190. package/src/core/colors.ts +63 -0
  191. package/src/core/index.ts +1206 -0
  192. package/src/core/keys.ts +63 -0
  193. package/src/core/types.ts +104 -0
  194. package/src/index.ts +171 -0
  195. package/src/math/index.ts +516 -0
  196. package/src/mobile/index.ts +294 -0
  197. package/src/models/index.ts +1258 -0
  198. package/src/physics/index.ts +1134 -0
  199. package/src/scene/index.ts +698 -0
  200. package/src/shapes/index.ts +120 -0
  201. package/src/text/index.ts +48 -0
  202. package/src/textures/index.ts +187 -0
  203. package/src/vfx/index.ts +191 -0
  204. package/src/world/index.ts +24 -0
  205. package/src/world/loader.ts +423 -0
  206. package/src/world/prefab.ts +217 -0
  207. package/src/world/render.ts +172 -0
  208. package/src/world/saver.ts +108 -0
  209. package/src/world/serialize.ts +301 -0
  210. package/src/world/terrain.ts +355 -0
  211. package/src/world/types.ts +160 -0
  212. package/src/world/validate.ts +319 -0
  213. package/src/world/version.ts +114 -0
@@ -0,0 +1,1586 @@
1
+ //! Screen-space GI probes and SSR (placement, trace SW/HW/SDF,
2
+ //! temporal, resolve). Split from renderer/shaders.rs.
3
+
4
+
5
+ // ============================================================================
6
+ // Ticket 007a — Lumen-style screen-probe SSGI (software Hi-Z trace)
7
+ //
8
+ // One probe per 16×16 half-res-pixel tile. Each probe stores 64 radiance
9
+ // samples in an 8×8 octahedral atlas. Passes: place → trace → temporal →
10
+ // resolve. The resolve pass writes the legacy `ssgi_rt` so downstream
11
+ // compositing is untouched.
12
+ // ============================================================================
13
+
14
+ /// Shared helpers prepended to every probe compute/fragment shader.
15
+ /// Contains octahedral encode/decode, view-space reconstruction, and
16
+ /// the Hi-Z sample helper. Kept as a Rust &str so it can be prepended
17
+ /// in the shader-module setup without a WGSL include mechanism.
18
+ pub(in crate::renderer) const PROBE_HELPERS_WGSL: &str = "
19
+ const PROBE_TILE_SIZE: u32 = 16u;
20
+ const PROBE_OCT_SIZE: u32 = 8u;
21
+ const PROBE_OCT_TEXELS: u32 = 64u;
22
+ const HIZ_SKY_Z: f32 = 10000.0;
23
+ const PI: f32 = 3.14159265;
24
+
25
+ struct ProbeHeader {
26
+ // xyz = world-space probe position; w = valid (1.0 = on surface, 0.0 = sky/invalid)
27
+ world_pos: vec4<f32>,
28
+ // xyz = world-space normal at the probe surface; w = linear |view-z|
29
+ normal: vec4<f32>,
30
+ };
31
+
32
+ fn oct_wrap(v: vec2<f32>) -> vec2<f32> {
33
+ let s = vec2<f32>(
34
+ select(-1.0, 1.0, v.x >= 0.0),
35
+ select(-1.0, 1.0, v.y >= 0.0),
36
+ );
37
+ return (1.0 - abs(vec2<f32>(v.y, v.x))) * s;
38
+ }
39
+
40
+ fn oct_encode(n_in: vec3<f32>) -> vec2<f32> {
41
+ let n = n_in / (abs(n_in.x) + abs(n_in.y) + abs(n_in.z));
42
+ let xy = select(oct_wrap(n.xy), n.xy, n.z >= 0.0);
43
+ return xy * 0.5 + 0.5;
44
+ }
45
+
46
+ fn oct_decode(uv: vec2<f32>) -> vec3<f32> {
47
+ let f = uv * 2.0 - 1.0;
48
+ var n = vec3<f32>(f.x, f.y, 1.0 - abs(f.x) - abs(f.y));
49
+ let t = max(-n.z, 0.0);
50
+ n.x = n.x + select(t, -t, n.x >= 0.0);
51
+ n.y = n.y + select(t, -t, n.y >= 0.0);
52
+ return normalize(n);
53
+ }
54
+
55
+ fn octel_direction(octel: vec2<u32>) -> vec3<f32> {
56
+ let uv = (vec2<f32>(octel) + vec2<f32>(0.5)) / f32(PROBE_OCT_SIZE);
57
+ return oct_decode(uv);
58
+ }
59
+
60
+ // Ticket 016 V1/V2 — temporal octahedral direction jitter with
61
+ // per-probe decorrelation (V2). V1 indexed a 2D R2 low-discrepancy
62
+ // sequence by frame, giving every probe the same per-frame sample
63
+ // offset. That means the 3×3 neighbourhood the resolve pass reads
64
+ // sees 9 probes sampling identical sub-texel positions — the
65
+ // spatial filter averages correlated noise, which is slower to
66
+ // converge than independent samples would be.
67
+ //
68
+ // V2 folds `probe_idx` into the sequence via a third low-
69
+ // discrepancy axis (1/g³ ≈ 0.4301597). Adjacent probes now land
70
+ // at different sub-texel positions each frame, so the 3×3 read
71
+ // effectively samples 9 × 4 = 36 distinct directions per octel
72
+ // over the EMA horizon rather than 4. Same zero-cost structure
73
+ // as V1 — two `fract` calls with an extra multiply.
74
+ const OCT_JITTER_A1: f32 = 0.7548776662;
75
+ const OCT_JITTER_A2: f32 = 0.5698402910;
76
+ const OCT_JITTER_A3: f32 = 0.4301597090;
77
+ fn octel_jitter(frame: f32, probe_idx: u32) -> vec2<f32> {
78
+ // R2 sequence in `frame` + orthogonal axis in `probe_idx`.
79
+ // The two components use different probe-axis scales (a3 vs
80
+ // a3 × R2's irrational) to stay 2D-decorrelated across probes.
81
+ let pf = f32(probe_idx);
82
+ return vec2<f32>(
83
+ fract(0.5 + OCT_JITTER_A1 * frame + OCT_JITTER_A3 * pf) - 0.5,
84
+ fract(0.5 + OCT_JITTER_A2 * frame + OCT_JITTER_A3 * pf * 1.324718) - 0.5,
85
+ );
86
+ }
87
+ fn octel_direction_jittered(octel: vec2<u32>, jitter: vec2<f32>) -> vec3<f32> {
88
+ let uv = (vec2<f32>(octel) + vec2<f32>(0.5) + jitter) / f32(PROBE_OCT_SIZE);
89
+ return oct_decode(uv);
90
+ }
91
+
92
+ fn view_pos_from_linear(uv: vec2<f32>, linear_z: f32,
93
+ p00: f32, p11: f32, p20: f32, p21: f32) -> vec3<f32> {
94
+ let ndc_x = uv.x * 2.0 - 1.0;
95
+ let ndc_y = 1.0 - uv.y * 2.0;
96
+ let view_z = -linear_z;
97
+ let view_x = -(ndc_x + p20) * view_z / p00;
98
+ let view_y = -(ndc_y + p21) * view_z / p11;
99
+ return vec3<f32>(view_x, view_y, view_z);
100
+ }
101
+
102
+ fn ign(p: vec2<f32>) -> f32 {
103
+ return fract(52.9829189 * fract(0.06711056 * p.x + 0.00583715 * p.y));
104
+ }
105
+ ";
106
+
107
+ /// Probe placement. One workgroup invocation per probe tile writes a
108
+ /// ProbeHeader (world position + world normal + linear view-z). Sky
109
+ /// probes are flagged invalid (world_pos.w = 0). Per-frame Halton-style
110
+ /// jitter moves the probe within its 16×16 tile so adjacent frames
111
+ /// cover slightly different surface points, widening effective
112
+ /// coverage when combined with temporal accumulation.
113
+ pub(in crate::renderer) const SSGI_PROBE_PLACE_WGSL: &str = "
114
+ struct PlaceParams {
115
+ // Full inverse view matrix — used to lift view-space positions/normals
116
+ // back into world space so the trace can march across the scene.
117
+ inv_view: mat4x4<f32>,
118
+ // x = proj[0][0], y = proj[1][1], z = proj[2][0], w = proj[2][1]
119
+ proj_row01: vec4<f32>,
120
+ // x = half_w, y = half_h, z = grid_w, w = grid_h
121
+ size: vec4<u32>,
122
+ // x = frame_index (temporal jitter), y = tile_size_f (16.0), zw unused
123
+ params: vec4<f32>,
124
+ };
125
+
126
+ @group(0) @binding(0) var<uniform> u: PlaceParams;
127
+ @group(0) @binding(1) var hiz0: texture_2d<f32>;
128
+ @group(0) @binding(2) var hiz_samp: sampler;
129
+ @group(0) @binding(3) var<storage, read_write> probes: array<ProbeHeader>;
130
+
131
+ @compute @workgroup_size(8, 8, 1)
132
+ fn cs_main(@builtin(global_invocation_id) gid: vec3<u32>) {
133
+ let grid_w = u.size.z;
134
+ let grid_h = u.size.w;
135
+ if (gid.x >= grid_w || gid.y >= grid_h) { return; }
136
+
137
+ let probe_idx = gid.y * grid_w + gid.x;
138
+ let half_w = f32(u.size.x);
139
+ let half_h = f32(u.size.y);
140
+ let tile = u.params.y;
141
+ let frame = u.params.x;
142
+
143
+ // Jitter UV inside the tile — 50% of tile radius so probe stays
144
+ // comfortably away from tile borders. Golden-ratio offsets across
145
+ // frames decorrelate the jitter from TAA/SSAO patterns.
146
+ let jx = ign(vec2<f32>(f32(gid.x) + frame * 1.618, f32(gid.y)));
147
+ let jy = ign(vec2<f32>(f32(gid.x), f32(gid.y) + frame * 2.236));
148
+ let px_x = f32(gid.x) * tile + tile * 0.5 + (jx - 0.5) * tile * 0.5;
149
+ let px_y = f32(gid.y) * tile + tile * 0.5 + (jy - 0.5) * tile * 0.5;
150
+ let uv = vec2<f32>(px_x / half_w, px_y / half_h);
151
+
152
+ let linear_z = textureSampleLevel(hiz0, hiz_samp, uv, 0.0).r;
153
+
154
+ // Sky probe — mark invalid and bail.
155
+ if (linear_z >= HIZ_SKY_Z * 0.5) {
156
+ probes[probe_idx].world_pos = vec4<f32>(0.0);
157
+ probes[probe_idx].normal = vec4<f32>(0.0, 1.0, 0.0, 0.0);
158
+ return;
159
+ }
160
+
161
+ let p00 = u.proj_row01.x;
162
+ let p11 = u.proj_row01.y;
163
+ let p20 = u.proj_row01.z;
164
+ let p21 = u.proj_row01.w;
165
+ let P = view_pos_from_linear(uv, linear_z, p00, p11, p20, p21);
166
+
167
+ // Finite-difference normal from 3-tap view-pos cross product. One
168
+ // texel to the right and one up. Uses the same Hi-Z mip 0 the
169
+ // center tap read from.
170
+ let texel = vec2<f32>(1.0 / half_w, 1.0 / half_h);
171
+ let uv_r = uv + vec2<f32>(texel.x, 0.0);
172
+ let uv_u = uv + vec2<f32>(0.0, -texel.y);
173
+ let zr = textureSampleLevel(hiz0, hiz_samp, uv_r, 0.0).r;
174
+ let zu = textureSampleLevel(hiz0, hiz_samp, uv_u, 0.0).r;
175
+ let P_r = view_pos_from_linear(uv_r, zr, p00, p11, p20, p21);
176
+ let P_u = view_pos_from_linear(uv_u, zu, p00, p11, p20, p21);
177
+ let N_vs = normalize(cross(P_r - P, P_u - P));
178
+
179
+ let P_world = (u.inv_view * vec4<f32>(P, 1.0)).xyz;
180
+ let N_world = normalize((u.inv_view * vec4<f32>(N_vs, 0.0)).xyz);
181
+
182
+ probes[probe_idx].world_pos = vec4<f32>(P_world, 1.0);
183
+ probes[probe_idx].normal = vec4<f32>(N_world, linear_z);
184
+ }
185
+ ";
186
+
187
+ /// Probe trace, software (Hi-Z) path.
188
+ ///
189
+ /// One workgroup per probe; each of the 64 lanes handles one octahedral
190
+ /// texel = one ray direction. Hemisphere-cull: rays below the probe's
191
+ /// tangent plane contribute zero (not visible from this surface
192
+ /// orientation). Surviving rays march the Hi-Z depth pyramid in view
193
+ /// space and sample the HDR buffer at hit. Misses contribute zero —
194
+ /// sky/off-screen handling is the compose pass's job downstream.
195
+ pub(in crate::renderer) const SSGI_PROBE_TRACE_SW_WGSL: &str = "
196
+ struct TraceParams {
197
+ view: mat4x4<f32>,
198
+ proj: mat4x4<f32>,
199
+ inv_view: mat4x4<f32>,
200
+ proj_row01: vec4<f32>,
201
+ // x = half_w, y = half_h, z = grid_w, w = grid_h
202
+ size: vec4<u32>,
203
+ // x = frame_index, y = intensity, z = max_march_t_world, w = firefly_cap
204
+ params: vec4<f32>,
205
+ // Ticket 014 V3/V6/V13 — rest of the shared `ProbeTraceParams`
206
+ // layout. Ignored by Hi-Z; present only so the shader struct
207
+ // size matches the host uniform buffer. V13 replaced the single
208
+ // `wsrc` vec4 with a 3-element cascade array (xyz = origin,
209
+ // w = extent).
210
+ sun_dir: vec4<f32>,
211
+ sun_color: vec4<f32>,
212
+ sky_color: vec4<f32>,
213
+ clipmap: vec4<f32>,
214
+ wsrc_cascades: array<vec4<f32>, 3>,
215
+ };
216
+
217
+ @group(0) @binding(0) var<uniform> u: TraceParams;
218
+ @group(0) @binding(1) var<storage, read> probes: array<ProbeHeader>;
219
+ @group(0) @binding(2) var hiz0: texture_2d<f32>;
220
+ @group(0) @binding(3) var hiz1: texture_2d<f32>;
221
+ @group(0) @binding(4) var hiz2: texture_2d<f32>;
222
+ @group(0) @binding(5) var hiz3: texture_2d<f32>;
223
+ @group(0) @binding(6) var hiz4: texture_2d<f32>;
224
+ @group(0) @binding(7) var hiz_samp: sampler;
225
+ @group(0) @binding(8) var hdr_tex: texture_2d<f32>;
226
+ @group(0) @binding(9) var hdr_samp: sampler;
227
+ @group(0) @binding(10) var radiance_out: texture_storage_3d<rgba16float, write>;
228
+ @group(0) @binding(11) var prev_history: texture_3d<f32>;
229
+
230
+ fn hiz_sample(uv: vec2<f32>, mip: i32) -> f32 {
231
+ switch (clamp(mip, 0, 4)) {
232
+ case 0: { return textureSampleLevel(hiz0, hiz_samp, uv, 0.0).r; }
233
+ case 1: { return textureSampleLevel(hiz1, hiz_samp, uv, 0.0).r; }
234
+ case 2: { return textureSampleLevel(hiz2, hiz_samp, uv, 0.0).r; }
235
+ case 3: { return textureSampleLevel(hiz3, hiz_samp, uv, 0.0).r; }
236
+ default: { return textureSampleLevel(hiz4, hiz_samp, uv, 0.0).r; }
237
+ }
238
+ }
239
+
240
+ @compute @workgroup_size(8, 8, 1)
241
+ fn cs_main(
242
+ @builtin(workgroup_id) wg: vec3<u32>,
243
+ @builtin(local_invocation_id) lid: vec3<u32>,
244
+ ) {
245
+ let grid_w = u.size.z;
246
+ let grid_h = u.size.w;
247
+ if (wg.x >= grid_w || wg.y >= grid_h) { return; }
248
+ if (lid.x >= PROBE_OCT_SIZE || lid.y >= PROBE_OCT_SIZE) { return; }
249
+
250
+ let probe_idx = wg.y * grid_w + wg.x;
251
+ let header = probes[probe_idx];
252
+
253
+ let dst_coord = vec3<i32>(i32(wg.x), i32(wg.y), i32(lid.y * PROBE_OCT_SIZE + lid.x));
254
+
255
+ // Invalid probe (sky tile) → contribute zero.
256
+ if (header.world_pos.w < 0.5) {
257
+ textureStore(radiance_out, dst_coord, vec4<f32>(0.0));
258
+ return;
259
+ }
260
+
261
+ // V1 — temporal jitter within each octel; 4-frame EMA turns
262
+ // this into free super-sampling.
263
+ // V2 — probe_idx folded into the jitter so neighbouring probes
264
+ // sample decorrelated sub-texel positions.
265
+ // V3 — scale jitter inversely with prev-frame luma at this octel:
266
+ // already-bright octels narrow their jitter (exploit / lock in
267
+ // the peak); dark octels keep full jitter (explore for new
268
+ // light). Luma is read from the prev-frame temporal-filtered
269
+ // history texture; `dst_coord` indexes the probe × octel slab
270
+ // identically between trace output and history.
271
+ let prev_slice = textureLoad(prev_history, dst_coord, 0).rgb;
272
+ let prev_luma = dot(prev_slice, vec3<f32>(0.2126, 0.7152, 0.0722));
273
+ let jitter_scale = mix(1.0, 0.3, clamp(prev_luma, 0.0, 1.0));
274
+ let jitter = octel_jitter(u.params.x, probe_idx) * jitter_scale;
275
+ let dir_ws = octel_direction_jittered(lid.xy, jitter);
276
+ let n_ws = header.normal.xyz;
277
+
278
+ // Hemisphere cull — rays pointing below the surface carry no diffuse contribution.
279
+ let ndotd = dot(dir_ws, n_ws);
280
+ if (ndotd <= 0.0) {
281
+ textureStore(radiance_out, dst_coord, vec4<f32>(0.0));
282
+ return;
283
+ }
284
+
285
+ // Trace in view space so the Hi-Z march lines up directly with the
286
+ // rasterized depth pyramid. Start origin at probe_pos + small normal
287
+ // offset to avoid self-intersection at shading point.
288
+ let origin_ws = header.world_pos.xyz + n_ws * 0.02;
289
+ let origin_vs = (u.view * vec4<f32>(origin_ws, 1.0)).xyz;
290
+ let dir_vs = normalize((u.view * vec4<f32>(dir_ws, 0.0)).xyz);
291
+
292
+ let p00 = u.proj_row01.x;
293
+ let p11 = u.proj_row01.y;
294
+ let p20 = u.proj_row01.z;
295
+ let p21 = u.proj_row01.w;
296
+
297
+ let max_t = u.params.z;
298
+ var t = 0.05;
299
+ let n_steps: i32 = 14;
300
+ let growth = pow(max_t / t, 1.0 / f32(n_steps));
301
+
302
+ var hit_color = vec3<f32>(0.0);
303
+ var prev_t = 0.0;
304
+
305
+ for (var s = 0; s < n_steps; s = s + 1) {
306
+ let pt_vs = origin_vs + dir_vs * t;
307
+ let clip = u.proj * vec4<f32>(pt_vs, 1.0);
308
+ let ndc = clip.xyz / clip.w;
309
+
310
+ // Off-screen — no hit possible, stop.
311
+ if (ndc.x < -1.0 || ndc.x > 1.0 || ndc.y < -1.0 || ndc.y > 1.0 || ndc.z < 0.0 || ndc.z > 1.0) {
312
+ break;
313
+ }
314
+
315
+ let ray_uv = vec2<f32>(ndc.x * 0.5 + 0.5, 1.0 - (ndc.y * 0.5 + 0.5));
316
+
317
+ // Pick the mip such that step footprint ≈ one mip texel. Longer
318
+ // steps sample coarser mips so the early-out fires at coarse
319
+ // resolution; only the last few steps hit mip 0.
320
+ let step_size = t - prev_t;
321
+ let mip = clamp(i32(floor(log2(max(step_size / 0.05, 1.0)))), 0, 4);
322
+
323
+ let scene_z = hiz_sample(ray_uv, mip);
324
+ // Hi-Z stores positive |view-z|. ray view-z is negative.
325
+ let ray_abs_z = -pt_vs.z;
326
+
327
+ // Have we marched behind a surface? The step tolerance scales
328
+ // with step size: the final tolerance (step_size * 2 + 0.1)
329
+ // lets coarse steps accept wider thickness, which matches the
330
+ // existing SSGI behaviour closely enough for V1.
331
+ if (ray_abs_z >= scene_z && scene_z < HIZ_SKY_Z * 0.5) {
332
+ // Refine against mip 0 to reject far-off misses.
333
+ let refined_z = hiz_sample(ray_uv, 0);
334
+ let thickness = abs(ray_abs_z - refined_z);
335
+ if (thickness < step_size * 2.0 + 0.1) {
336
+ let tn = t / max_t;
337
+ let falloff = 1.0 - tn * tn;
338
+ var raw = textureSampleLevel(hdr_tex, hdr_samp, ray_uv, 0.0).rgb * max(falloff, 0.0);
339
+ // Firefly clamp (cap per-sample luma).
340
+ let luma = dot(raw, vec3<f32>(0.2126, 0.7152, 0.0722));
341
+ let cap = u.params.w;
342
+ if (luma > cap) { raw = raw * (cap / luma); }
343
+ hit_color = raw;
344
+ }
345
+ break;
346
+ }
347
+
348
+ prev_t = t;
349
+ t = t * growth;
350
+ }
351
+
352
+ let intensity = u.params.y;
353
+ textureStore(radiance_out, dst_coord, vec4<f32>(hit_color * intensity * ndotd, 1.0));
354
+ }
355
+ ";
356
+
357
+ /// Probe trace, hardware (ray-query) path (ticket 007b).
358
+ ///
359
+ /// Same workgroup shape as the SW shader — one workgroup per probe, 64
360
+ /// lanes per probe, each handling one octahedral texel. The per-ray
361
+ /// inner loop replaces Hi-Z screen-space marching with `rayQuery`
362
+ /// against the TLAS, which pulls off-screen geometry into the bounce
363
+ /// (the whole point of HW-RT here).
364
+ ///
365
+ /// Hit shading is "hit-lighting-lite":
366
+ /// - flat per-instance albedo + world-space normal from
367
+ /// `instance_data[hit.instance_custom_data]`;
368
+ /// - sun direct: NdotL × sun_color (no cascade shadow lookup in V1 —
369
+ /// the bias is one bounce away and hidden by temporal averaging;
370
+ /// re-add when shadow-aware hit shading becomes worth the cost);
371
+ /// - sky: max(dot(N, up), 0) × sky_color for the upward hemisphere;
372
+ /// - emissive: per-instance scalar × albedo;
373
+ /// - distance falloff and firefly clamp match the SW path so the
374
+ /// two trace variants are visually interchangeable where they
375
+ /// both see on-screen geometry.
376
+ pub(in crate::renderer) const SSGI_PROBE_TRACE_HW_WGSL: &str = "
377
+ struct TraceParams {
378
+ view: mat4x4<f32>,
379
+ proj: mat4x4<f32>,
380
+ inv_view: mat4x4<f32>,
381
+ proj_row01: vec4<f32>,
382
+ size: vec4<u32>,
383
+ params: vec4<f32>,
384
+ sun_dir: vec4<f32>,
385
+ sun_color: vec4<f32>,
386
+ sky_color: vec4<f32>,
387
+ // Ticket 014 V3/V6/V13 — clipmap + WSRC cascade array. HW path
388
+ // consumes `wsrc_cascades` on its miss branch; the clipmap field
389
+ // is padding here (HW ray-query has its own world-space trace).
390
+ clipmap: vec4<f32>,
391
+ wsrc_cascades: array<vec4<f32>, 3>,
392
+ };
393
+
394
+ struct InstanceGiData {
395
+ albedo: vec3<f32>,
396
+ emissive_luma: f32,
397
+ normal_ws: vec3<f32>,
398
+ _pad0: f32,
399
+ // Ticket 013 V2: x = first_slot_index (first of 6 consecutive
400
+ // signed-axis slots), yz unused, w = has_card flag.
401
+ card_slot: vec4<f32>,
402
+ // Object-space AABB min (xyz) / max (xyz).
403
+ card_aabb_min: vec4<f32>,
404
+ card_aabb_max: vec4<f32>,
405
+ // EN-023 — world-space AABB (SDF path only; layout mirror).
406
+ world_aabb_min: vec4<f32>,
407
+ world_aabb_max: vec4<f32>,
408
+ // PT-2 — layout mirror only (path-tracer geometry window +
409
+ // material params); the GI traces ignore both fields.
410
+ geo: vec4<u32>,
411
+ mat_params: vec4<f32>,
412
+ };
413
+
414
+ const CARD_SLOTS_PER_ROW: f32 = 64.0;
415
+ const HW_WSRC_GRID_RES: i32 = 16;
416
+
417
+ @group(0) @binding(0) var<uniform> u: TraceParams;
418
+ @group(0) @binding(1) var<storage, read> probes: array<ProbeHeader>;
419
+ @group(0) @binding(2) var accel: acceleration_structure;
420
+ @group(0) @binding(3) var<storage, read> instance_data: array<InstanceGiData>;
421
+ @group(0) @binding(4) var radiance_out: texture_storage_3d<rgba16float, write>;
422
+ @group(0) @binding(5) var card_atlas: texture_2d<f32>;
423
+ @group(0) @binding(6) var card_samp: sampler;
424
+ @group(0) @binding(7) var wsrc_atlas: texture_3d<f32>;
425
+ @group(0) @binding(8) var wsrc_samp: sampler;
426
+ @group(0) @binding(9) var prev_history: texture_3d<f32>;
427
+
428
+ // Ticket 014 V7/V8 — WSRC lookup shared with the SDF path. V8
429
+ // trilinear across the 8 neighbouring probes, nearest octel for
430
+ // direction. extent=0 is the cache-not-ready sentinel that the
431
+ // host writes before the first bake completes, so the HW miss
432
+ // falls back to the pre-V7 return-black behaviour.
433
+ // Ticket 014 V10/V13 — HW mirror of the SDF sampler-based WSRC
434
+ // lookup. Same 48-slice cascade packing + smallest-containing-
435
+ // cascade selection.
436
+ fn hw_wsrc_sample_probe(cascade: i32, gx: i32, gy: i32, gz_f: f32, ru: vec2<f32>) -> vec3<f32> {
437
+ let gxc = clamp(gx, 0, 15);
438
+ let gyc = clamp(gy, 0, 15);
439
+ let ax = (f32(gxc) + 0.1 + ru.x * 0.8) / 16.0;
440
+ let ay = (f32(gyc) + 0.1 + ru.y * 0.8) / 16.0;
441
+ let az = (f32(cascade) * 16.0 + gz_f) / 48.0;
442
+ return textureSampleLevel(wsrc_atlas, wsrc_samp,
443
+ vec3<f32>(ax, ay, az), 0.0).rgb;
444
+ }
445
+
446
+ fn hw_wsrc_pick_cascade(pos_ws: vec3<f32>) -> i32 {
447
+ for (var c: i32 = 0; c < 3; c = c + 1) {
448
+ let origin = u.wsrc_cascades[c].xyz;
449
+ let extent = u.wsrc_cascades[c].w;
450
+ if (extent <= 0.0) { continue; }
451
+ let rel = pos_ws - origin;
452
+ let half = extent * 0.5;
453
+ if (abs(rel.x) < half && abs(rel.y) < half && abs(rel.z) < half) {
454
+ return c;
455
+ }
456
+ }
457
+ return -1;
458
+ }
459
+
460
+ fn hw_wsrc_sample(pos_ws: vec3<f32>, dir_ws: vec3<f32>) -> vec3<f32> {
461
+ let cascade = hw_wsrc_pick_cascade(pos_ws);
462
+ if (cascade < 0) {
463
+ return vec3<f32>(0.0);
464
+ }
465
+ let origin = u.wsrc_cascades[cascade].xyz;
466
+ let extent = u.wsrc_cascades[cascade].w;
467
+ let cell = extent / 16.0;
468
+ let rel = pos_ws - origin + vec3<f32>(extent * 0.5);
469
+ let pf = rel / cell - vec3<f32>(0.5);
470
+ let pfx = floor(pf.x);
471
+ let pfy = floor(pf.y);
472
+ let gix = i32(pfx);
473
+ let giy = i32(pfy);
474
+ let fx = pf.x - pfx;
475
+ let fy = pf.y - pfy;
476
+ let gz_f = clamp(pf.z + 0.5, 0.5, 15.5);
477
+
478
+ let ru = oct_encode(dir_ws);
479
+
480
+ let c00 = hw_wsrc_sample_probe(cascade, gix, giy, gz_f, ru);
481
+ let c10 = hw_wsrc_sample_probe(cascade, gix + 1, giy, gz_f, ru);
482
+ let c01 = hw_wsrc_sample_probe(cascade, gix, giy + 1, gz_f, ru);
483
+ let c11 = hw_wsrc_sample_probe(cascade, gix + 1, giy + 1, gz_f, ru);
484
+
485
+ let ix = 1.0 - fx;
486
+ let iy = 1.0 - fy;
487
+ return c00 * (ix * iy) + c10 * (fx * iy)
488
+ + c01 * (ix * fy) + c11 * (fx * fy);
489
+ }
490
+
491
+ @compute @workgroup_size(8, 8, 1)
492
+ fn cs_main(
493
+ @builtin(workgroup_id) wg: vec3<u32>,
494
+ @builtin(local_invocation_id) lid: vec3<u32>,
495
+ ) {
496
+ let grid_w = u.size.z;
497
+ let grid_h = u.size.w;
498
+ if (wg.x >= grid_w || wg.y >= grid_h) { return; }
499
+ if (lid.x >= PROBE_OCT_SIZE || lid.y >= PROBE_OCT_SIZE) { return; }
500
+
501
+ let probe_idx = wg.y * grid_w + wg.x;
502
+ let header = probes[probe_idx];
503
+
504
+ let dst_coord = vec3<i32>(i32(wg.x), i32(wg.y), i32(lid.y * PROBE_OCT_SIZE + lid.x));
505
+
506
+ if (header.world_pos.w < 0.5) {
507
+ textureStore(radiance_out, dst_coord, vec4<f32>(0.0));
508
+ return;
509
+ }
510
+
511
+ // V1 — temporal jitter within each octel; 4-frame EMA turns
512
+ // this into free super-sampling.
513
+ // V2 — probe_idx folded into the jitter so neighbouring probes
514
+ // sample decorrelated sub-texel positions.
515
+ // V3 — scale jitter inversely with prev-frame luma at this octel:
516
+ // already-bright octels narrow their jitter (exploit / lock in
517
+ // the peak); dark octels keep full jitter (explore for new
518
+ // light). Luma is read from the prev-frame temporal-filtered
519
+ // history texture; `dst_coord` indexes the probe × octel slab
520
+ // identically between trace output and history.
521
+ let prev_slice = textureLoad(prev_history, dst_coord, 0).rgb;
522
+ let prev_luma = dot(prev_slice, vec3<f32>(0.2126, 0.7152, 0.0722));
523
+ let jitter_scale = mix(1.0, 0.3, clamp(prev_luma, 0.0, 1.0));
524
+ let jitter = octel_jitter(u.params.x, probe_idx) * jitter_scale;
525
+ let dir_ws = octel_direction_jittered(lid.xy, jitter);
526
+ let n_ws = header.normal.xyz;
527
+ let ndotd = dot(dir_ws, n_ws);
528
+ if (ndotd <= 0.0) {
529
+ textureStore(radiance_out, dst_coord, vec4<f32>(0.0));
530
+ return;
531
+ }
532
+
533
+ // 2 cm normal offset — matches the SW start_t and keeps primary
534
+ // hits from self-intersecting the surface the probe sits on.
535
+ let origin_ws = header.world_pos.xyz + n_ws * 0.02;
536
+ let max_t = u.params.z;
537
+
538
+ var rq: ray_query;
539
+ rayQueryInitialize(&rq, accel, RayDesc(
540
+ 0u,
541
+ 0xFFu,
542
+ 0.001,
543
+ max_t,
544
+ origin_ws,
545
+ dir_ws,
546
+ ));
547
+ loop {
548
+ if (!rayQueryProceed(&rq)) { break; }
549
+ }
550
+ let hit = rayQueryGetCommittedIntersection(&rq);
551
+
552
+ var radiance = vec3<f32>(0.0);
553
+ if (hit.kind != RAY_QUERY_INTERSECTION_NONE) {
554
+ let inst = instance_data[hit.instance_custom_data];
555
+
556
+ // Ticket 013 V2 — pick the axis facing the incoming ray and
557
+ // sample the pre-lit radiance atlas. Each mesh has 6
558
+ // consecutive slots laid out by signed axis (see host-side
559
+ // capture loop). The card's world-space normal was baked at
560
+ // capture into `card_slot_meta`; lighting was applied once
561
+ // per frame by `card_light_pass`, so the sample IS the
562
+ // bounce contribution — no hit-time shading math.
563
+ if (inst.card_slot.w > 0.5) {
564
+ let hit_world = origin_ws + dir_ws * hit.t;
565
+ let hit_os = (hit.world_to_object * vec4<f32>(hit_world, 1.0)).xyz;
566
+ // Dominant component of the world-space ray direction
567
+ // picks the major axis; its sign selects the front/back
568
+ // face of that axis.
569
+ let abs_d = abs(dir_ws);
570
+ var axis_idx: u32 = 0u;
571
+ if (abs_d.y >= abs_d.x && abs_d.y >= abs_d.z) {
572
+ axis_idx = 2u;
573
+ } else if (abs_d.z >= abs_d.x) {
574
+ axis_idx = 4u;
575
+ }
576
+ // Sign: ray going -X (dir.x < 0) lands on +X face → axis + 0.
577
+ // ray going +X lands on -X face → axis + 1.
578
+ var signed_axis: u32 = axis_idx;
579
+ if (axis_idx == 0u && dir_ws.x > 0.0) { signed_axis = 1u; }
580
+ else if (axis_idx == 2u && dir_ws.y > 0.0) { signed_axis = 3u; }
581
+ else if (axis_idx == 4u && dir_ws.z > 0.0) { signed_axis = 5u; }
582
+
583
+ let first_slot = u32(inst.card_slot.x);
584
+ let slot = first_slot + signed_axis;
585
+ let slot_x = slot % 64u;
586
+ let slot_y = slot / 64u;
587
+
588
+ // Project hit_os onto the card plane — same math as V1 but
589
+ // with signed-axis-aware u sign flips so the ±pair cards
590
+ // pick up opposite views of the mesh.
591
+ let bmin = inst.card_aabb_min.xyz;
592
+ let bmax = inst.card_aabb_max.xyz;
593
+ var u_os: f32;
594
+ var v_os: f32;
595
+ var u_lo: f32;
596
+ var u_hi: f32;
597
+ var v_lo: f32;
598
+ var v_hi: f32;
599
+ var u_flip: f32 = 1.0;
600
+ if (signed_axis == 0u || signed_axis == 1u) {
601
+ u_os = hit_os.y; v_os = hit_os.z;
602
+ u_lo = bmin.y; u_hi = bmax.y; v_lo = bmin.z; v_hi = bmax.z;
603
+ if (signed_axis == 1u) { u_flip = -1.0; }
604
+ } else if (signed_axis == 2u || signed_axis == 3u) {
605
+ u_os = hit_os.x; v_os = hit_os.z;
606
+ u_lo = bmin.x; u_hi = bmax.x; v_lo = bmin.z; v_hi = bmax.z;
607
+ if (signed_axis == 3u) { u_flip = -1.0; }
608
+ } else {
609
+ u_os = hit_os.x; v_os = hit_os.y;
610
+ u_lo = bmin.x; u_hi = bmax.x; v_lo = bmin.y; v_hi = bmax.y;
611
+ if (signed_axis == 5u) { u_flip = -1.0; }
612
+ }
613
+ var u_norm = clamp((u_os - u_lo) / max(u_hi - u_lo, 1e-4), 0.0, 1.0);
614
+ let v_norm = clamp((v_os - v_lo) / max(v_hi - v_lo, 1e-4), 0.0, 1.0);
615
+ if (u_flip < 0.0) { u_norm = 1.0 - u_norm; }
616
+
617
+ let slot_size_uv = 1.0 / CARD_SLOTS_PER_ROW;
618
+ let texel_in_slot = slot_size_uv / f32(64); // 64×64 card
619
+ let slot_u0 = f32(slot_x) * slot_size_uv + texel_in_slot;
620
+ let slot_v0 = f32(slot_y) * slot_size_uv + texel_in_slot;
621
+ let slot_span = slot_size_uv - 2.0 * texel_in_slot;
622
+ let atlas_uv = vec2<f32>(
623
+ slot_u0 + u_norm * slot_span,
624
+ slot_v0 + v_norm * slot_span,
625
+ );
626
+ let pre_lit = textureSampleLevel(card_atlas, card_samp, atlas_uv, 0.0).rgb;
627
+
628
+ // V3 — emissive is pre-added into the radiance atlas by the
629
+ // card-lighting pass, so the hit simply picks up the full
630
+ // pre-lit texel and applies distance falloff + firefly cap.
631
+ let tn = hit.t / max_t;
632
+ let falloff = max(1.0 - tn * tn, 0.0);
633
+ var raw = pre_lit * falloff;
634
+ let luma = dot(raw, vec3<f32>(0.2126, 0.7152, 0.0722));
635
+ let cap = u.params.w;
636
+ if (luma > cap) { raw = raw * (cap / luma); }
637
+ radiance = raw;
638
+ } else {
639
+ // Fallback path — no card captured, use the flat
640
+ // instance albedo × hit-time lighting same as 007b.
641
+ let hit_n = inst.normal_ws;
642
+ let ndotl = max(dot(hit_n, u.sun_dir.xyz), 0.0);
643
+ let direct = u.sun_color.xyz * ndotl;
644
+ let ndotup = max(dot(hit_n, vec3<f32>(0.0, 1.0, 0.0)), 0.0);
645
+ let sky = u.sky_color.xyz * ndotup;
646
+ let tn = hit.t / max_t;
647
+ let falloff = max(1.0 - tn * tn, 0.0);
648
+ var raw = inst.albedo * (direct + sky) * falloff
649
+ + inst.albedo * inst.emissive_luma;
650
+ let luma = dot(raw, vec3<f32>(0.2126, 0.7152, 0.0722));
651
+ let cap = u.params.w;
652
+ if (luma > cap) { raw = raw * (cap / luma); }
653
+ radiance = raw;
654
+ }
655
+ } else {
656
+ // Ticket 014 V7 — miss path samples the WSRC envelope so HW
657
+ // traces that escape scene geometry still contribute sky /
658
+ // sun-visibility signal. Terminal position is the ray's full
659
+ // march distance; direction picks the octel on the nearest
660
+ // probe.
661
+ let terminal = origin_ws + dir_ws * max_t;
662
+ var raw = hw_wsrc_sample(terminal, dir_ws);
663
+ let luma = dot(raw, vec3<f32>(0.2126, 0.7152, 0.0722));
664
+ let cap = u.params.w;
665
+ if (luma > cap) { raw = raw * (cap / luma); }
666
+ radiance = raw;
667
+ }
668
+
669
+ let intensity = u.params.y;
670
+ textureStore(radiance_out, dst_coord, vec4<f32>(radiance * intensity * ndotd, 1.0));
671
+ }
672
+ ";
673
+
674
+ /// Ticket 014 V3 — probe trace, software SDF sphere-march path.
675
+ ///
676
+ /// Third trace variant alongside the 007a Hi-Z screen-space and 007b
677
+ /// HW ray-query paths. Same workgroup shape (8×8 = one workgroup per
678
+ /// probe, each lane handles one octahedral texel). Only the per-ray
679
+ /// inner loop differs: instead of marching screen-space depth or
680
+ /// firing a `rayQuery`, each lane sphere-marches the scene-wide SDF
681
+ /// clipmap baked by ticket 014 V2.
682
+ ///
683
+ /// Hit shading is intentionally minimal for V3 — no mesh-card lookup
684
+ /// (the clipmap is a merged SDF with no per-instance identity). At
685
+ /// hit we estimate the surface normal by finite-differencing the SDF
686
+ /// clipmap around the hit point, then apply analytic sun × NdotL +
687
+ /// sky × NdotUp against a constant gray albedo. That gives SW-only
688
+ /// adapters a working one-bounce indirect — lower quality than the
689
+ /// 013 Mesh-Cards HW path but self-contained.
690
+ ///
691
+ /// The clipmap is a single R32Float 3D texture covering a fixed world-
692
+ /// space AABB defined at capture (see `SCENE_SDF_CLIPMAP_EXTENT` /
693
+ /// `SCENE_SDF_CLIPMAP_ORIGIN` on the Rust side). Rays whose marched
694
+ /// position leaves the AABB treat the miss as "open sky".
695
+ pub(in crate::renderer) const SSGI_PROBE_TRACE_SDF_WGSL: &str = "
696
+ struct TraceParams {
697
+ view: mat4x4<f32>,
698
+ proj: mat4x4<f32>,
699
+ inv_view: mat4x4<f32>,
700
+ proj_row01: vec4<f32>,
701
+ size: vec4<u32>,
702
+ // x = frame_index, y = intensity, z = max_march_t, w = firefly_cap
703
+ params: vec4<f32>,
704
+ sun_dir: vec4<f32>,
705
+ sun_color: vec4<f32>,
706
+ sky_color: vec4<f32>,
707
+ // xyz = clipmap origin, w = extent (full width, not half)
708
+ clipmap: vec4<f32>,
709
+ // Ticket 014 V6/V13 — WSRC cascade cubes. Each element is
710
+ // (origin xyz, extent w). Cascades are ordered near→far; the
711
+ // miss path picks the smallest cascade whose cube contains the
712
+ // ray-terminal position. extent <= 0 marks an unbaked cascade
713
+ // (per-cascade); the shader falls back to black if none match.
714
+ wsrc_cascades: array<vec4<f32>, 3>,
715
+ };
716
+
717
+ struct SdfInstanceGiData {
718
+ albedo: vec3<f32>,
719
+ emissive_luma: f32,
720
+ normal_ws: vec3<f32>,
721
+ _pad0: f32,
722
+ card_slot: vec4<f32>,
723
+ card_aabb_min: vec4<f32>,
724
+ card_aabb_max: vec4<f32>,
725
+ // EN-023 — world-space AABB. This trace marches a WORLD-space
726
+ // clipmap and has no world_to_object; comparing world hits against
727
+ // the object-space box above only ever worked for identity-
728
+ // transform assets (Sponza).
729
+ world_aabb_min: vec4<f32>,
730
+ world_aabb_max: vec4<f32>,
731
+ };
732
+
733
+ const SDF_CARD_SLOTS_PER_ROW: f32 = 64.0;
734
+ const SDF_CARD_SLOT_PX: u32 = 64u;
735
+ const WSRC_GRID_RES: i32 = 16;
736
+
737
+ @group(0) @binding(0) var<uniform> u: TraceParams;
738
+ @group(0) @binding(1) var<storage, read> probes: array<ProbeHeader>;
739
+ @group(0) @binding(2) var clipmap_tex: texture_3d<f32>;
740
+ @group(0) @binding(3) var clipmap_samp: sampler;
741
+ @group(0) @binding(4) var radiance_out: texture_storage_3d<rgba16float, write>;
742
+ @group(0) @binding(5) var<storage, read> instance_data: array<SdfInstanceGiData>;
743
+ @group(0) @binding(6) var card_atlas: texture_2d<f32>;
744
+ @group(0) @binding(7) var card_samp: sampler;
745
+ @group(0) @binding(8) var wsrc_atlas: texture_3d<f32>;
746
+ @group(0) @binding(9) var wsrc_samp: sampler;
747
+ @group(0) @binding(10) var prev_history: texture_3d<f32>;
748
+
749
+ fn clipmap_uv(pos_ws: vec3<f32>) -> vec3<f32> {
750
+ let half_extent = u.clipmap.w * 0.5;
751
+ let origin = u.clipmap.xyz;
752
+ return (pos_ws - origin + vec3<f32>(half_extent)) / u.clipmap.w;
753
+ }
754
+
755
+ fn clipmap_sample(pos_ws: vec3<f32>) -> f32 {
756
+ let uv = clipmap_uv(pos_ws);
757
+ if (uv.x < 0.0 || uv.x > 1.0 || uv.y < 0.0 || uv.y > 1.0 || uv.z < 0.0 || uv.z > 1.0) {
758
+ // Outside the clipmap — assume wide open (no hit possible).
759
+ return 1e4;
760
+ }
761
+ return textureSampleLevel(clipmap_tex, clipmap_samp, uv, 0.0).r;
762
+ }
763
+
764
+ // Ticket 014 V10/V13 — WSRC lookup via the hardware linear-filtering
765
+ // sampler, now multi-cascade. Each cascade occupies 16 z-slices of
766
+ // the atlas at depth offset `cascade_idx * 16`. The miss path picks
767
+ // the smallest cascade whose cube contains `pos_ws` and does the
768
+ // V10 4-sample trilinear inside that cascade.
769
+ //
770
+ // Atlas packing (per cascade, same within each 16-slice block):
771
+ // probe (gx, gy, gz) at padded octel (ox_p, oy_p in [0, 9]) lives
772
+ // at texel `(gx*10 + ox_p, gy*10 + oy_p, cascade * 16 + gz)`.
773
+ // Real octel sits at padded (ox+1, oy+1). Borders are
774
+ // octahedrally-wrapped at bake (V11).
775
+ //
776
+ // Sampler uv formula (atlas x-axis): `atlas_uv_x = (gx + 0.1 +
777
+ // ru_x * 0.8) / 16`. Z picks the cascade: `atlas_uv_z = (c * 16 +
778
+ // gz + 0.5 + fz) / 48` for 3 cascades.
779
+ fn wsrc_sample_probe(cascade: i32, gx: i32, gy: i32, gz_f: f32, ru: vec2<f32>) -> vec3<f32> {
780
+ let gxc = clamp(gx, 0, 15);
781
+ let gyc = clamp(gy, 0, 15);
782
+ let ax = (f32(gxc) + 0.1 + ru.x * 0.8) / 16.0;
783
+ let ay = (f32(gyc) + 0.1 + ru.y * 0.8) / 16.0;
784
+ // 48-slice atlas = 3 cascades × 16 probes in Z. Sample at the
785
+ // cascade's slice block; the `gz_f` carries the per-cascade
786
+ // sub-slice fraction (already centred for the sampler).
787
+ let az = (f32(cascade) * 16.0 + gz_f) / 48.0;
788
+ return textureSampleLevel(wsrc_atlas, wsrc_samp,
789
+ vec3<f32>(ax, ay, az), 0.0).rgb;
790
+ }
791
+
792
+ // V13 — pick the first cascade whose cube contains `pos_ws` and is
793
+ // built (extent > 0). Returns -1 if none match.
794
+ fn wsrc_pick_cascade(pos_ws: vec3<f32>) -> i32 {
795
+ for (var c: i32 = 0; c < 3; c = c + 1) {
796
+ let origin = u.wsrc_cascades[c].xyz;
797
+ let extent = u.wsrc_cascades[c].w;
798
+ if (extent <= 0.0) { continue; }
799
+ let rel = pos_ws - origin;
800
+ let half = extent * 0.5;
801
+ if (abs(rel.x) < half && abs(rel.y) < half && abs(rel.z) < half) {
802
+ return c;
803
+ }
804
+ }
805
+ return -1;
806
+ }
807
+
808
+ fn wsrc_sample(pos_ws: vec3<f32>, dir_ws: vec3<f32>) -> vec3<f32> {
809
+ let cascade = wsrc_pick_cascade(pos_ws);
810
+ if (cascade < 0) {
811
+ return vec3<f32>(0.0);
812
+ }
813
+ let origin = u.wsrc_cascades[cascade].xyz;
814
+ let extent = u.wsrc_cascades[cascade].w;
815
+ let cell = extent / 16.0;
816
+ let rel = pos_ws - origin + vec3<f32>(extent * 0.5);
817
+ let pf = rel / cell - vec3<f32>(0.5);
818
+ let pfx = floor(pf.x);
819
+ let pfy = floor(pf.y);
820
+ let gix = i32(pfx);
821
+ let giy = i32(pfy);
822
+ let fx = pf.x - pfx;
823
+ let fy = pf.y - pfy;
824
+ let gz_f = clamp(pf.z + 0.5, 0.5, 15.5);
825
+
826
+ let ru = oct_encode(dir_ws);
827
+
828
+ let c00 = wsrc_sample_probe(cascade, gix, giy, gz_f, ru);
829
+ let c10 = wsrc_sample_probe(cascade, gix + 1, giy, gz_f, ru);
830
+ let c01 = wsrc_sample_probe(cascade, gix, giy + 1, gz_f, ru);
831
+ let c11 = wsrc_sample_probe(cascade, gix + 1, giy + 1, gz_f, ru);
832
+
833
+ let ix = 1.0 - fx;
834
+ let iy = 1.0 - fy;
835
+ return c00 * (ix * iy) + c10 * (fx * iy)
836
+ + c01 * (ix * fy) + c11 * (fx * fy);
837
+ }
838
+
839
+ @compute @workgroup_size(8, 8, 1)
840
+ fn cs_main(
841
+ @builtin(workgroup_id) wg: vec3<u32>,
842
+ @builtin(local_invocation_id) lid: vec3<u32>,
843
+ ) {
844
+ let grid_w = u.size.z;
845
+ let grid_h = u.size.w;
846
+ if (wg.x >= grid_w || wg.y >= grid_h) { return; }
847
+ if (lid.x >= PROBE_OCT_SIZE || lid.y >= PROBE_OCT_SIZE) { return; }
848
+
849
+ let probe_idx = wg.y * grid_w + wg.x;
850
+ let header = probes[probe_idx];
851
+ let dst_coord = vec3<i32>(i32(wg.x), i32(wg.y), i32(lid.y * PROBE_OCT_SIZE + lid.x));
852
+
853
+ if (header.world_pos.w < 0.5) {
854
+ textureStore(radiance_out, dst_coord, vec4<f32>(0.0));
855
+ return;
856
+ }
857
+
858
+ // V1 — temporal jitter within each octel; 4-frame EMA turns
859
+ // this into free super-sampling.
860
+ // V2 — probe_idx folded into the jitter so neighbouring probes
861
+ // sample decorrelated sub-texel positions.
862
+ // V3 — scale jitter inversely with prev-frame luma at this octel:
863
+ // already-bright octels narrow their jitter (exploit / lock in
864
+ // the peak); dark octels keep full jitter (explore for new
865
+ // light). Luma is read from the prev-frame temporal-filtered
866
+ // history texture; `dst_coord` indexes the probe × octel slab
867
+ // identically between trace output and history.
868
+ let prev_slice = textureLoad(prev_history, dst_coord, 0).rgb;
869
+ let prev_luma = dot(prev_slice, vec3<f32>(0.2126, 0.7152, 0.0722));
870
+ let jitter_scale = mix(1.0, 0.3, clamp(prev_luma, 0.0, 1.0));
871
+ let jitter = octel_jitter(u.params.x, probe_idx) * jitter_scale;
872
+ let dir_ws = octel_direction_jittered(lid.xy, jitter);
873
+ let n_ws = header.normal.xyz;
874
+ let ndotd = dot(dir_ws, n_ws);
875
+ if (ndotd <= 0.0) {
876
+ textureStore(radiance_out, dst_coord, vec4<f32>(0.0));
877
+ return;
878
+ }
879
+
880
+ // 2 cm normal offset matches the SW Hi-Z + HW ray-query paths —
881
+ // keeps primary hits from self-intersecting the probe surface.
882
+ let origin_ws = header.world_pos.xyz + n_ws * 0.02;
883
+ let max_t = u.params.z;
884
+
885
+ // Sphere-trace. Step is the UDF value; convergence when within a
886
+ // voxel's worth of the surface or when we exhaust the budget.
887
+ let voxel_size = u.clipmap.w / 64.0; // 64³ clipmap resolution
888
+ let hit_threshold = voxel_size * 1.5;
889
+ var t: f32 = 0.0;
890
+ var hit: bool = false;
891
+ for (var s: i32 = 0; s < 48; s = s + 1) {
892
+ let pos = origin_ws + dir_ws * t;
893
+ let d = clipmap_sample(pos);
894
+ if (d < hit_threshold) {
895
+ hit = true;
896
+ break;
897
+ }
898
+ t = t + max(d, voxel_size * 0.5);
899
+ if (t >= max_t) { break; }
900
+ }
901
+
902
+ var radiance = vec3<f32>(0.0);
903
+ if (hit) {
904
+ let hit_pos = origin_ws + dir_ws * t;
905
+
906
+ // UDF gradient → outward surface normal (flip since gradient
907
+ // points AWAY from the surface in an unsigned field).
908
+ let h = voxel_size;
909
+ let dx = clipmap_sample(hit_pos + vec3<f32>(h, 0.0, 0.0))
910
+ - clipmap_sample(hit_pos - vec3<f32>(h, 0.0, 0.0));
911
+ let dy = clipmap_sample(hit_pos + vec3<f32>(0.0, h, 0.0))
912
+ - clipmap_sample(hit_pos - vec3<f32>(0.0, h, 0.0));
913
+ let dz = clipmap_sample(hit_pos + vec3<f32>(0.0, 0.0, h))
914
+ - clipmap_sample(hit_pos - vec3<f32>(0.0, 0.0, h));
915
+ var grad = vec3<f32>(dx, dy, dz);
916
+ let glen = length(grad);
917
+ if (glen > 1e-4) { grad = grad / glen; }
918
+ let hit_n = -grad;
919
+
920
+ // Ticket 014 V4 — broad-phase lookup: walk `instance_data`,
921
+ // find the first AABB (slightly dilated) containing hit_pos.
922
+ // Pick the axis most aligned with the outward normal; project
923
+ // hit onto its card; sample the pre-lit radiance atlas. Falls
924
+ // back to analytic sun/sky × gray when no instance matches
925
+ // (clipmap sentinel voxels, hits inside unaccounted-for
926
+ // geometry, etc.).
927
+ let count = arrayLength(&instance_data);
928
+ var picked: i32 = -1;
929
+ var picked_vol: f32 = 1e30;
930
+ for (var i: u32 = 0u; i < count; i = i + 1u) {
931
+ let ad = instance_data[i];
932
+ if (ad.card_slot.w < 0.5) { continue; }
933
+ // EN-023 — compare the WORLD hit against the WORLD AABB.
934
+ // The old object-space comparison only matched assets whose
935
+ // vertices were already in world space; every transformed
936
+ // instance fell through to the gray analytic fallback.
937
+ // Pick the SMALLEST containing box, not the first: a scene-
938
+ // spanning instance (the shooter's ±140 m terrain proxy)
939
+ // otherwise swallows every hit — walls and trees included —
940
+ // and its mostly-empty side cards darken the bounce.
941
+ let bmin = ad.world_aabb_min.xyz - vec3<f32>(0.05);
942
+ let bmax = ad.world_aabb_max.xyz + vec3<f32>(0.05);
943
+ if (hit_pos.x >= bmin.x && hit_pos.x <= bmax.x &&
944
+ hit_pos.y >= bmin.y && hit_pos.y <= bmax.y &&
945
+ hit_pos.z >= bmin.z && hit_pos.z <= bmax.z) {
946
+ let ext = bmax - bmin;
947
+ let vol = ext.x * ext.y * ext.z;
948
+ if (vol < picked_vol) {
949
+ picked = i32(i);
950
+ picked_vol = vol;
951
+ }
952
+ }
953
+ }
954
+
955
+ if (picked >= 0) {
956
+ let ad = instance_data[u32(picked)];
957
+ // Pick signed axis from outward normal. Dominant component
958
+ // picks the axis; sign picks + or - face.
959
+ let abs_n = abs(hit_n);
960
+ var axis_idx: u32 = 0u;
961
+ if (abs_n.y >= abs_n.x && abs_n.y >= abs_n.z) {
962
+ axis_idx = 2u;
963
+ } else if (abs_n.z >= abs_n.x) {
964
+ axis_idx = 4u;
965
+ }
966
+ var signed_axis: u32 = axis_idx;
967
+ if (axis_idx == 0u && hit_n.x < 0.0) { signed_axis = 1u; }
968
+ else if (axis_idx == 2u && hit_n.y < 0.0) { signed_axis = 3u; }
969
+ else if (axis_idx == 4u && hit_n.z < 0.0) { signed_axis = 5u; }
970
+
971
+ let first_slot = u32(ad.card_slot.x);
972
+ let slot = first_slot + signed_axis;
973
+ let slot_x = slot % 64u;
974
+ let slot_y = slot / 64u;
975
+
976
+ // EN-023 — project against the WORLD AABB, consistent with
977
+ // the world-space hit. Exact for the translate+scale (yaw-0)
978
+ // instances the GI proxies use; a yaw-rotated instance would
979
+ // sample its card with rotated UVs — hue still right, which
980
+ // is what the probe integral actually consumes at 64² cards.
981
+ let bmin = ad.world_aabb_min.xyz;
982
+ let bmax = ad.world_aabb_max.xyz;
983
+ var u_os: f32;
984
+ var v_os: f32;
985
+ var u_lo: f32; var u_hi: f32;
986
+ var v_lo: f32; var v_hi: f32;
987
+ var u_flip: f32 = 1.0;
988
+ if (signed_axis == 0u || signed_axis == 1u) {
989
+ u_os = hit_pos.y; v_os = hit_pos.z;
990
+ u_lo = bmin.y; u_hi = bmax.y; v_lo = bmin.z; v_hi = bmax.z;
991
+ if (signed_axis == 1u) { u_flip = -1.0; }
992
+ } else if (signed_axis == 2u || signed_axis == 3u) {
993
+ u_os = hit_pos.x; v_os = hit_pos.z;
994
+ u_lo = bmin.x; u_hi = bmax.x; v_lo = bmin.z; v_hi = bmax.z;
995
+ if (signed_axis == 3u) { u_flip = -1.0; }
996
+ } else {
997
+ u_os = hit_pos.x; v_os = hit_pos.y;
998
+ u_lo = bmin.x; u_hi = bmax.x; v_lo = bmin.y; v_hi = bmax.y;
999
+ if (signed_axis == 5u) { u_flip = -1.0; }
1000
+ }
1001
+ var u_norm = clamp((u_os - u_lo) / max(u_hi - u_lo, 1e-4), 0.0, 1.0);
1002
+ let v_norm = clamp((v_os - v_lo) / max(v_hi - v_lo, 1e-4), 0.0, 1.0);
1003
+ if (u_flip < 0.0) { u_norm = 1.0 - u_norm; }
1004
+ let slot_size_uv = 1.0 / SDF_CARD_SLOTS_PER_ROW;
1005
+ let texel_in_slot = slot_size_uv / f32(SDF_CARD_SLOT_PX);
1006
+ let slot_u0 = f32(slot_x) * slot_size_uv + texel_in_slot;
1007
+ let slot_v0 = f32(slot_y) * slot_size_uv + texel_in_slot;
1008
+ let slot_span = slot_size_uv - 2.0 * texel_in_slot;
1009
+ let atlas_uv = vec2<f32>(
1010
+ slot_u0 + u_norm * slot_span,
1011
+ slot_v0 + v_norm * slot_span,
1012
+ );
1013
+ let pre_lit = textureSampleLevel(card_atlas, card_samp, atlas_uv, 0.0).rgb;
1014
+
1015
+ let tn = t / max_t;
1016
+ let falloff = max(1.0 - tn * tn, 0.0);
1017
+ var raw = pre_lit * falloff;
1018
+ let luma = dot(raw, vec3<f32>(0.2126, 0.7152, 0.0722));
1019
+ let cap = u.params.w;
1020
+ if (luma > cap) { raw = raw * (cap / luma); }
1021
+ radiance = raw;
1022
+ } else {
1023
+ // Fallback — analytic sun/sky × gray albedo when no
1024
+ // instance matches. Same shading as V3.
1025
+ let ndotl = max(dot(hit_n, u.sun_dir.xyz), 0.0);
1026
+ let direct = u.sun_color.xyz * ndotl;
1027
+ let ndotup = max(dot(hit_n, vec3<f32>(0.0, 1.0, 0.0)), 0.0);
1028
+ let sky = u.sky_color.xyz * ndotup;
1029
+ let albedo = vec3<f32>(0.55, 0.55, 0.55);
1030
+ let tn = t / max_t;
1031
+ let falloff = max(1.0 - tn * tn, 0.0);
1032
+ var raw = albedo * (direct + sky) * falloff;
1033
+ let luma = dot(raw, vec3<f32>(0.2126, 0.7152, 0.0722));
1034
+ let cap = u.params.w;
1035
+ if (luma > cap) { raw = raw * (cap / luma); }
1036
+ radiance = raw;
1037
+ }
1038
+ } else {
1039
+ // Ticket 014 V6 — miss path samples the WSRC envelope instead
1040
+ // of returning black. Ray terminal position (origin + dir * t)
1041
+ // is where we project into the cache; direction picks the
1042
+ // probe's octel. Firefly-clamp to match the hit path.
1043
+ let terminal = origin_ws + dir_ws * t;
1044
+ var raw = wsrc_sample(terminal, dir_ws);
1045
+ let luma = dot(raw, vec3<f32>(0.2126, 0.7152, 0.0722));
1046
+ let cap = u.params.w;
1047
+ if (luma > cap) { raw = raw * (cap / luma); }
1048
+ radiance = raw;
1049
+ }
1050
+
1051
+ let intensity = u.params.y;
1052
+ textureStore(radiance_out, dst_coord, vec4<f32>(radiance * intensity * ndotd, 1.0));
1053
+ }
1054
+ ";
1055
+
1056
+ /// Probe temporal accumulator. EMA in probe-octel space. No reprojection
1057
+ /// in V1 — since every frame traces all 64 octels, history only smooths
1058
+ /// firefly noise; per-probe world positions jitter by tile-fraction so
1059
+ /// camera motion eventually converges to a stable signal without explicit
1060
+ /// velocity. Disocclusion is handled implicitly: moving the camera
1061
+ /// changes which tile a surface falls into, replacing that probe's
1062
+ /// header, and the new probe's history blends from zero since the old
1063
+ /// probe at that grid coord pointed elsewhere.
1064
+ pub(in crate::renderer) const SSGI_PROBE_TEMPORAL_WGSL: &str = "
1065
+ struct TemporalParams {
1066
+ // x = alpha (0.25 = 4-frame EMA at steady state),
1067
+ // y = force_refresh (1 → alpha 1.0),
1068
+ // z = grid_w, w = grid_h
1069
+ params: vec4<f32>,
1070
+ };
1071
+
1072
+ @group(0) @binding(0) var<uniform> u: TemporalParams;
1073
+ @group(0) @binding(1) var radiance_in: texture_3d<f32>;
1074
+ @group(0) @binding(2) var history_in: texture_3d<f32>;
1075
+ @group(0) @binding(3) var history_out: texture_storage_3d<rgba16float, write>;
1076
+
1077
+ @compute @workgroup_size(8, 8, 1)
1078
+ fn cs_main(
1079
+ @builtin(workgroup_id) wg: vec3<u32>,
1080
+ @builtin(local_invocation_id) lid: vec3<u32>,
1081
+ ) {
1082
+ let grid_w = u32(u.params.z);
1083
+ let grid_h = u32(u.params.w);
1084
+ if (wg.x >= grid_w || wg.y >= grid_h) { return; }
1085
+
1086
+ let coord = vec3<i32>(i32(wg.x), i32(wg.y), i32(lid.y * PROBE_OCT_SIZE + lid.x));
1087
+ let curr = textureLoad(radiance_in, coord, 0).rgb;
1088
+ let hist = textureLoad(history_in, coord, 0).rgb;
1089
+
1090
+ var alpha = u.params.x;
1091
+ if (u.params.y > 0.5) {
1092
+ alpha = 1.0;
1093
+ } else {
1094
+ // Ticket 016 V4 — variance-adaptive alpha. Scale the base
1095
+ // EMA by `|luma(curr) - luma(hist)|` so moving lights /
1096
+ // disocclusions / scene cuts converge quickly while stable
1097
+ // octels keep strong temporal smoothing. This captures the
1098
+ // hierarchical-refinement intent (high-variance regions get
1099
+ // more per-frame weight, low-variance regions average more
1100
+ // history) without needing a separate refinement probe
1101
+ // layer + indirect dispatch.
1102
+ //
1103
+ // `luma_delta_scale = 0.6` means a 1.0-luma delta pushes
1104
+ // alpha up by 0.6 on top of the 0.25 base — up to 0.85
1105
+ // before the `min(1.0)` clamp.
1106
+ let curr_luma = dot(curr, vec3<f32>(0.2126, 0.7152, 0.0722));
1107
+ let hist_luma = dot(hist, vec3<f32>(0.2126, 0.7152, 0.0722));
1108
+ let delta = abs(curr_luma - hist_luma);
1109
+ alpha = min(1.0, alpha + delta * 0.6);
1110
+ }
1111
+ let blended = mix(hist, curr, alpha);
1112
+
1113
+ textureStore(history_out, coord, vec4<f32>(blended, 1.0));
1114
+ }
1115
+ ";
1116
+
1117
+ /// Per-pixel probe-cache reconstruction. Writes the half-res ssgi_rt
1118
+ /// that the downstream compose / TAA passes already read.
1119
+ ///
1120
+ /// Samples the 2×2 probes whose tiles enclose the pixel's tile. For
1121
+ /// each probe, evaluates the octahedral atlas along the pixel's
1122
+ /// world-space normal, then bilateral-weights the contribution by
1123
+ /// depth-match + normal-match with the pixel itself. Invalid probes
1124
+ /// (sky) are skipped. When all 4 probes reject (pixel depth/normal
1125
+ /// wildly off), fall back to a zero contribution — better than leaking
1126
+ /// a stale distant probe's radiance into a foreground surface.
1127
+ pub(in crate::renderer) const SSGI_PROBE_RESOLVE_WGSL: &str = "
1128
+ struct ResolveParams {
1129
+ inv_view: mat4x4<f32>,
1130
+ proj_row01: vec4<f32>,
1131
+ // x = half_w, y = half_h, z = grid_w, w = grid_h
1132
+ size: vec4<u32>,
1133
+ // x = tile_size (16.0), y = intensity, zw unused
1134
+ params: vec4<f32>,
1135
+ };
1136
+
1137
+ @group(0) @binding(0) var<uniform> u: ResolveParams;
1138
+ @group(0) @binding(1) var<storage, read> probes: array<ProbeHeader>;
1139
+ @group(0) @binding(2) var radiance_tex: texture_3d<f32>;
1140
+ @group(0) @binding(3) var radiance_samp: sampler;
1141
+ @group(0) @binding(4) var hiz0: texture_2d<f32>;
1142
+ @group(0) @binding(5) var hiz_samp: sampler;
1143
+
1144
+ struct VsOut {
1145
+ @builtin(position) clip_pos: vec4<f32>,
1146
+ @location(0) uv: vec2<f32>,
1147
+ };
1148
+
1149
+ @vertex
1150
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
1151
+ let x = f32((vid & 1u) * 4u) - 1.0;
1152
+ let y = f32((vid >> 1u) * 4u) - 1.0;
1153
+ var out: VsOut;
1154
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
1155
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
1156
+ return out;
1157
+ }
1158
+
1159
+ // Sample a probe's octahedral atlas in a given world-space direction.
1160
+ // Uses trilinear sampling on the 3D texture so neighbouring octels
1161
+ // softly blend — some visible smear on the 8×8 atlas, at the cost of
1162
+ // a cheap reconstruction.
1163
+ fn sample_probe(probe_xy: vec2<i32>, dir_ws: vec3<f32>) -> vec3<f32> {
1164
+ let oct_uv = oct_encode(dir_ws);
1165
+ let u_tex = (f32(probe_xy.x) + 0.5) / f32(u.size.z);
1166
+ let v_tex = (f32(probe_xy.y) + 0.5) / f32(u.size.w);
1167
+ // z coordinate: octahedral texel index in [0, 64) normalized to [0, 1)
1168
+ let oct_x = clamp(oct_uv.x * f32(PROBE_OCT_SIZE), 0.0, f32(PROBE_OCT_SIZE) - 1.0);
1169
+ let oct_y = clamp(oct_uv.y * f32(PROBE_OCT_SIZE), 0.0, f32(PROBE_OCT_SIZE) - 1.0);
1170
+ let z_idx = floor(oct_y) * f32(PROBE_OCT_SIZE) + floor(oct_x);
1171
+ let z = (z_idx + 0.5) / f32(PROBE_OCT_TEXELS);
1172
+ return textureSampleLevel(radiance_tex, radiance_samp, vec3<f32>(u_tex, v_tex, z), 0.0).rgb;
1173
+ }
1174
+
1175
+ @fragment
1176
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
1177
+ let linear_z = textureSampleLevel(hiz0, hiz_samp, in.uv, 0.0).r;
1178
+ if (linear_z >= HIZ_SKY_Z * 0.5) {
1179
+ return vec4<f32>(0.0);
1180
+ }
1181
+
1182
+ let half_w = f32(u.size.x);
1183
+ let half_h = f32(u.size.y);
1184
+ let tile = u.params.x;
1185
+ let grid_w = i32(u.size.z);
1186
+ let grid_h = i32(u.size.w);
1187
+
1188
+ let p00 = u.proj_row01.x;
1189
+ let p11 = u.proj_row01.y;
1190
+ let p20 = u.proj_row01.z;
1191
+ let p21 = u.proj_row01.w;
1192
+ let P_vs = view_pos_from_linear(in.uv, linear_z, p00, p11, p20, p21);
1193
+
1194
+ // Reconstruct pixel normal (same 3-tap trick as the placement pass).
1195
+ let texel = vec2<f32>(1.0 / half_w, 1.0 / half_h);
1196
+ let zr = textureSampleLevel(hiz0, hiz_samp, in.uv + vec2<f32>(texel.x, 0.0), 0.0).r;
1197
+ let zu = textureSampleLevel(hiz0, hiz_samp, in.uv + vec2<f32>(0.0, -texel.y), 0.0).r;
1198
+ let Pr = view_pos_from_linear(in.uv + vec2<f32>(texel.x, 0.0), zr, p00, p11, p20, p21);
1199
+ let Pu = view_pos_from_linear(in.uv + vec2<f32>(0.0, -texel.y), zu, p00, p11, p20, p21);
1200
+ let N_vs = normalize(cross(Pr - P_vs, Pu - P_vs));
1201
+ let N_ws = normalize((u.inv_view * vec4<f32>(N_vs, 0.0)).xyz);
1202
+
1203
+ // Pixel's grid-space fractional position (which probes surround it?).
1204
+ let px_x = in.uv.x * half_w;
1205
+ let px_y = in.uv.y * half_h;
1206
+ let fx = px_x / tile - 0.5; // -0.5 aligns grid cells centred on tile centres
1207
+ let fy = px_y / tile - 0.5;
1208
+ let gx0 = i32(floor(fx));
1209
+ let gy0 = i32(floor(fy));
1210
+ let tx = fract(fx);
1211
+ let ty = fract(fy);
1212
+
1213
+ var accum = vec3<f32>(0.0);
1214
+ var wsum = 0.0;
1215
+
1216
+ for (var dy = 0; dy <= 1; dy = dy + 1) {
1217
+ for (var dx = 0; dx <= 1; dx = dx + 1) {
1218
+ let gx = clamp(gx0 + dx, 0, grid_w - 1);
1219
+ let gy = clamp(gy0 + dy, 0, grid_h - 1);
1220
+ let probe = probes[u32(gy * grid_w + gx)];
1221
+ if (probe.world_pos.w < 0.5) { continue; }
1222
+
1223
+ // Bilinear corner weight
1224
+ var w_corner = 1.0;
1225
+ w_corner = w_corner * select(1.0 - tx, tx, dx == 1);
1226
+ w_corner = w_corner * select(1.0 - ty, ty, dy == 1);
1227
+
1228
+ // Depth + normal bilateral weights — reject probes on very
1229
+ // different surfaces from the pixel (foreground pixel vs
1230
+ // probe on a far wall, or on an orthogonal facet).
1231
+ let dz = abs(probe.normal.w - linear_z);
1232
+ let w_depth = exp(-dz * dz * 8.0);
1233
+ let ndotn = clamp(dot(probe.normal.xyz, N_ws), 0.0, 1.0);
1234
+ let w_normal = pow(ndotn, 4.0);
1235
+
1236
+ let w = w_corner * w_depth * w_normal;
1237
+ if (w <= 0.0001) { continue; }
1238
+
1239
+ let radiance = sample_probe(vec2<i32>(gx, gy), N_ws);
1240
+ accum = accum + radiance * w;
1241
+ wsum = wsum + w;
1242
+ }
1243
+ }
1244
+
1245
+ if (wsum > 0.0001) {
1246
+ accum = (accum / wsum) * u.params.y;
1247
+ }
1248
+ return vec4<f32>(accum, 1.0);
1249
+ }
1250
+ ";
1251
+
1252
+ /// SSR temporal denoiser. Same shape as the SSGI temporal pass:
1253
+ /// reprojects the previous history through the motion vectors,
1254
+ /// clamps against the 3×3 neighborhood of the noisy current frame,
1255
+ /// and blends with a low alpha so 4–8 frames of random GGX rays
1256
+ /// converge to a smooth reflection. Also pre-filters the noisy
1257
+ /// current frame by the 3×3 mean, which kills single-pixel
1258
+ /// glossy-ray sparkles in one frame instead of 10.
1259
+ pub(in crate::renderer) const SSR_TEMPORAL_SHADER_WGSL: &str = "
1260
+ struct SsrTemporalParams {
1261
+ /// x = blend_alpha (0.1), yzw unused
1262
+ params: vec4<f32>,
1263
+ };
1264
+
1265
+ @group(0) @binding(0) var<uniform> u: SsrTemporalParams;
1266
+ @group(0) @binding(1) var current_tex: texture_2d<f32>;
1267
+ @group(0) @binding(2) var current_samp: sampler;
1268
+ @group(0) @binding(3) var history_tex: texture_2d<f32>;
1269
+ @group(0) @binding(4) var history_samp: sampler;
1270
+ @group(0) @binding(5) var velocity_tex: texture_2d<f32>;
1271
+ @group(0) @binding(6) var velocity_samp: sampler;
1272
+
1273
+ struct VsOut {
1274
+ @builtin(position) clip_pos: vec4<f32>,
1275
+ @location(0) uv: vec2<f32>,
1276
+ };
1277
+
1278
+ @vertex
1279
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
1280
+ let x = f32((vid & 1u) * 4u) - 1.0;
1281
+ let y = f32((vid >> 1u) * 4u) - 1.0;
1282
+ var out: VsOut;
1283
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
1284
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
1285
+ return out;
1286
+ }
1287
+
1288
+ @fragment
1289
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
1290
+ let current_raw = textureSampleLevel(current_tex, current_samp, in.uv, 0.0);
1291
+
1292
+ // 3×3 box pre-filter + neighborhood min/max. One texel spread
1293
+ // across 9 samples hides single-pixel glossy-ray sparkles in a
1294
+ // single frame; the min/max bounds the history so disocclusion
1295
+ // and material transitions clamp rather than ghost.
1296
+ let texel = vec2<f32>(1.0) / vec2<f32>(textureDimensions(current_tex));
1297
+ var nmin = current_raw;
1298
+ var nmax = current_raw;
1299
+ var prefilt = vec4<f32>(0.0);
1300
+ for (var y = -1; y <= 1; y++) {
1301
+ for (var x = -1; x <= 1; x++) {
1302
+ let s = textureSampleLevel(current_tex, current_samp, in.uv + vec2<f32>(f32(x), f32(y)) * texel, 0.0);
1303
+ nmin = min(nmin, s);
1304
+ nmax = max(nmax, s);
1305
+ prefilt = prefilt + s;
1306
+ }
1307
+ }
1308
+ let current = prefilt * (1.0 / 9.0);
1309
+
1310
+ // Velocity is full-res; UV mapping handles the half-res delta.
1311
+ // NDC-space velocity + UV Y-flip → `uv + vel.y` for the Y axis,
1312
+ // matching TAA + SSAO + the sibling SSGI temporal pass.
1313
+ let vel = textureSampleLevel(velocity_tex, velocity_samp, in.uv, 0.0).xy;
1314
+ let prev_uv = vec2<f32>(in.uv.x - vel.x, in.uv.y + vel.y);
1315
+ let off_screen = prev_uv.x < 0.0 || prev_uv.x > 1.0 || prev_uv.y < 0.0 || prev_uv.y > 1.0;
1316
+ if (off_screen) { return current; }
1317
+
1318
+ let history_raw = textureSampleLevel(history_tex, history_samp, prev_uv, 0.0);
1319
+ // Scrub NaN/Inf from the history read. Until a clean SSR frame
1320
+ // finishes draining the ping-pong pair, any poisoned history
1321
+ // pixel would otherwise survive the clamp (clamp(NaN, a, b) is
1322
+ // implementation-defined on Metal — frequently NaN) and keep
1323
+ // tonemapping to pink. Replace poisoned channels with the
1324
+ // current-frame mean, which is the best available estimate.
1325
+ let history = select(current, history_raw, history_raw == history_raw);
1326
+ let clamped_history = clamp(history, nmin, nmax);
1327
+ let alpha = u.params.x;
1328
+ let blended = mix(clamped_history, current, alpha);
1329
+ return select(current, blended, blended == blended);
1330
+ }
1331
+ ";
1332
+
1333
+ /// SSR (screen-space reflections) shader. View-space ray march:
1334
+ ///
1335
+ /// 1. Reconstruct view-space position from the depth buffer.
1336
+ /// 2. Reconstruct view-space normal from depth derivatives
1337
+ /// (cross of dpdx/dpdy of view position).
1338
+ /// 3. Reflect view direction around N → reflection direction R.
1339
+ /// 4. March along R in view space, project each step to screen
1340
+ /// coords, sample depth there, hit if our marched z is past the
1341
+ /// sampled surface.
1342
+ /// 5. On hit, sample the HDR RT at the hit UV and output it
1343
+ /// (faded toward edges of screen so off-screen reflections
1344
+ /// don't pop into existence).
1345
+ ///
1346
+ /// Output is half-res HDR. The TAA pass adds it on top of the
1347
+ /// prefiltered IBL specular for the final image.
1348
+ pub(in crate::renderer) const SSR_SHADER_WGSL: &str = "
1349
+ struct SsrParams {
1350
+ /// Inverse of the projection matrix — depth → view-space pos.
1351
+ inv_proj: mat4x4<f32>,
1352
+ /// Projection matrix — view-space pos → clip-space.
1353
+ proj: mat4x4<f32>,
1354
+ /// x = SSR strength (0 = off, 1 = full)
1355
+ /// y = max march distance in view-space units
1356
+ /// z = number of march steps
1357
+ /// w = frame index (Hammersley rotation + march jitter)
1358
+ params: vec4<f32>,
1359
+ /// EN-021 — view→world ROTATION (inverse of the view matrix's 3×3)
1360
+ /// so the env-miss fallback can turn the view-space reflection ray
1361
+ /// into a world direction for the equirect lookup.
1362
+ inv_view_rot: mat4x4<f32>,
1363
+ /// EN-021 — x = env max LOD (matches the material path's
1364
+ /// roughness×6 mip ramp), y = env intensity, zw unused.
1365
+ params2: vec4<f32>,
1366
+ };
1367
+
1368
+ @group(0) @binding(0) var<uniform> u: SsrParams;
1369
+ @group(0) @binding(1) var depth_tex: texture_depth_2d;
1370
+ @group(0) @binding(2) var depth_samp: sampler;
1371
+ @group(0) @binding(3) var hdr_tex: texture_2d<f32>;
1372
+ @group(0) @binding(4) var hdr_samp: sampler;
1373
+ @group(0) @binding(5) var mat_tex: texture_2d<f32>;
1374
+ @group(0) @binding(6) var mat_samp: sampler;
1375
+ @group(0) @binding(7) var albedo_tex: texture_2d<f32>;
1376
+ @group(0) @binding(8) var albedo_samp: sampler;
1377
+ @group(0) @binding(9) var env_tex: texture_2d<f32>;
1378
+ @group(0) @binding(10) var env_samp: sampler;
1379
+
1380
+ const PI: f32 = 3.14159265;
1381
+
1382
+ struct VsOut {
1383
+ @builtin(position) clip_pos: vec4<f32>,
1384
+ @location(0) uv: vec2<f32>,
1385
+ };
1386
+
1387
+ @vertex
1388
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
1389
+ let x = f32((vid & 1u) * 4u) - 1.0;
1390
+ let y = f32((vid >> 1u) * 4u) - 1.0;
1391
+ var out: VsOut;
1392
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
1393
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
1394
+ return out;
1395
+ }
1396
+
1397
+ fn view_pos_from_depth(uv: vec2<f32>, depth: f32) -> vec3<f32> {
1398
+ let ndc = vec4<f32>(uv.x * 2.0 - 1.0, (1.0 - uv.y) * 2.0 - 1.0, depth, 1.0);
1399
+ let view_h = u.inv_proj * ndc;
1400
+ return view_h.xyz / view_h.w;
1401
+ }
1402
+
1403
+ /// EN-021 — env-miss fallback. The scene shader scales its IBL specular
1404
+ /// down by SSR's ownership share (lighting.dir_light_count.z × the same
1405
+ /// roughness fade this shader uses), so a miss MUST return the env
1406
+ /// sample instead of black or off-screen reflections go dark. Same
1407
+ /// equirect mapping as common/pbr.wgsl's sample_env; explicit-LOD
1408
+ /// sampling needs no seam handling.
1409
+ fn env_fallback(r_view: vec3<f32>, roughness: f32) -> vec3<f32> {
1410
+ let d = normalize((u.inv_view_rot * vec4<f32>(r_view, 0.0)).xyz);
1411
+ let theta = acos(clamp(d.y, -1.0, 1.0));
1412
+ let phi = atan2(d.z, d.x);
1413
+ let uu = phi / (2.0 * PI);
1414
+ let uv = vec2<f32>(uu - floor(uu), theta / PI);
1415
+ return textureSampleLevel(env_tex, env_samp, uv, roughness * u.params2.x).rgb
1416
+ * u.params2.y;
1417
+ }
1418
+
1419
+ /// Interleaved gradient noise — per-pixel pseudo-random in [0, 1).
1420
+ /// Varies with frame so the temporal accumulator averages over
1421
+ /// different march offsets each frame.
1422
+ fn ign_jitter(frag_coord: vec2<f32>, frame: f32) -> f32 {
1423
+ let shifted = frag_coord + vec2<f32>(frame * 5.588238, frame * 3.127137);
1424
+ return fract(52.9829189 * fract(0.06711056 * shifted.x + 0.00583715 * shifted.y));
1425
+ }
1426
+
1427
+ /// Cheap 2D hash → two independent low-discrepancy values in [0,1)².
1428
+ /// Used as GGX microfacet-sample coordinates; the frame index rotates
1429
+ /// the hash so each pixel draws a different sample every frame and
1430
+ /// the temporal denoiser averages over the GGX lobe.
1431
+ fn hash2(frag_coord: vec2<f32>, frame: f32) -> vec2<f32> {
1432
+ let p1 = frag_coord + vec2<f32>(frame * 11.13, frame * 7.77);
1433
+ let p2 = frag_coord + vec2<f32>(frame * 3.17, frame * 5.29);
1434
+ let a = fract(sin(dot(p1, vec2<f32>(12.9898, 78.233))) * 43758.5453);
1435
+ let b = fract(sin(dot(p2, vec2<f32>(37.7191, 17.1123))) * 28471.1713);
1436
+ return vec2<f32>(a, b);
1437
+ }
1438
+
1439
+ /// GGX importance-sampled microfacet half-vector in tangent space
1440
+ /// aligned to the surface normal. Isotropic GGX — α = roughness².
1441
+ fn importance_sample_ggx(xi: vec2<f32>, n: vec3<f32>, roughness: f32) -> vec3<f32> {
1442
+ let a = roughness * roughness;
1443
+ let phi = 2.0 * PI * xi.x;
1444
+ let cos_theta = sqrt((1.0 - xi.y) / (1.0 + (a * a - 1.0) * xi.y));
1445
+ let sin_theta = sqrt(max(1.0 - cos_theta * cos_theta, 0.0));
1446
+ let h_local = vec3<f32>(sin_theta * cos(phi), sin_theta * sin(phi), cos_theta);
1447
+ let up = select(vec3<f32>(1.0, 0.0, 0.0), vec3<f32>(0.0, 0.0, 1.0), abs(n.z) < 0.999);
1448
+ let t = normalize(cross(up, n));
1449
+ let b = cross(n, t);
1450
+ return normalize(t * h_local.x + b * h_local.y + n * h_local.z);
1451
+ }
1452
+
1453
+ @fragment
1454
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
1455
+ let depth = textureSampleLevel(depth_tex, depth_samp, in.uv, 0i);
1456
+ // Derivatives must come from uniform control flow (WGSL uniformity
1457
+ // analysis; Tint/WebGPU enforces what naga lets slide) — take them
1458
+ // BEFORE the early returns. A few ALU on sky pixels, and the quad
1459
+ // derivatives are actually well-defined now instead of reading
1460
+ // helper-lane garbage at the sky/geometry boundary.
1461
+ let view_pos = view_pos_from_depth(in.uv, depth);
1462
+ let dx = dpdx(view_pos);
1463
+ let dy = dpdy(view_pos);
1464
+ if (depth >= 0.9999) { return vec4<f32>(0.0); } // sky
1465
+
1466
+ // SSR for every smooth-enough surface — the metals-only gate is gone.
1467
+ // EN-021 EXCLUSIVE OWNERSHIP: within this shader's roughness_fade
1468
+ // range, SSR owns specular reflections outright — the scene shader
1469
+ // scales its IBL specular by the complement of the same fade
1470
+ // (× strength, piped through lighting.dir_light_count.z), and a
1471
+ // march MISS falls back to the env sample here instead of black.
1472
+ // Metals previously double-counted on hit (IBL spec in hdr + full
1473
+ // SSR on top, round-2 audit F10); dielectrics were starved of IBL
1474
+ // spec by design and now get the coherent hit-or-env behaviour too.
1475
+ // Very rough surfaces still fade to pure IBL where one-ray-per-pixel
1476
+ // SSR noise would dominate even after temporal accumulation.
1477
+ let mat = textureSampleLevel(mat_tex, mat_samp, in.uv, 0.0).rg;
1478
+ let metallic = mat.r;
1479
+ let roughness = mat.g;
1480
+ let albedo = textureSampleLevel(albedo_tex, albedo_samp, in.uv, 0.0).rgb;
1481
+ let roughness_fade = 1.0 - smoothstep(0.5, 0.85, roughness);
1482
+ if (roughness_fade <= 0.001) { return vec4<f32>(0.0); }
1483
+
1484
+ let n = normalize(cross(dx, dy));
1485
+ let v = normalize(-view_pos);
1486
+
1487
+ // Stochastic SSR — cast one GGX-importance-sampled ray per pixel
1488
+ // per frame. Different frames draw from different points on the
1489
+ // GGX lobe (rotated by frame index) so the downstream temporal
1490
+ // denoiser averages a dense roughness cone over 4–8 frames. This
1491
+ // replaces the 5-tap Gaussian blur at the hit: we pay one ray +
1492
+ // one hdr sample per frame, not 32 march steps + 5 blur taps.
1493
+ //
1494
+ // xi is clamped away from exact 0 and 1. At roughness → 0 the GGX
1495
+ // denominator `1 + (α²-1)·xi.y` collapses to 0 when xi.y = 1, so
1496
+ // cos_theta becomes sqrt(0/0) = NaN. That NaN then propagates into
1497
+ // ssr_history and each 4× upsampled texel turns into a pink hot
1498
+ // pixel after tonemapping. Sponza's mirror-smooth lamp fittings
1499
+ // (roughness near 0) are exactly the worst case.
1500
+ let xi = clamp(hash2(in.clip_pos.xy, u.params.w), vec2<f32>(1e-4), vec2<f32>(0.9999));
1501
+ let h = importance_sample_ggx(xi, n, roughness);
1502
+ let r = reflect(-v, h);
1503
+
1504
+ let n_dot_v = max(dot(n, v), 0.0);
1505
+ let f0 = mix(vec3<f32>(0.04), albedo, metallic);
1506
+ let fresnel = f0 + (vec3<f32>(1.0) - f0) * pow(1.0 - n_dot_v, 5.0);
1507
+
1508
+ // Camera-facing rays can't be marched — env fallback (EN-021).
1509
+ if (r.z > 0.0) {
1510
+ let fb = env_fallback(r, roughness) * fresnel * roughness_fade * u.params.x;
1511
+ let fb_safe = select(vec3<f32>(0.0), fb, fb == fb);
1512
+ return vec4<f32>(fb_safe, 0.0);
1513
+ }
1514
+
1515
+ let max_dist = u.params.y;
1516
+ let n_steps_f = u.params.z;
1517
+ let n_steps = u32(n_steps_f);
1518
+ let step_size = max_dist / n_steps_f;
1519
+
1520
+ let jitter = ign_jitter(in.clip_pos.xy, u.params.w);
1521
+ var t = step_size * (0.5 + jitter);
1522
+
1523
+ var hit_uv = vec2<f32>(-1.0);
1524
+ var hit_found = false;
1525
+ var prev_t = 0.0;
1526
+ // FXC (the legacy HLSL compiler used by D3D11 + DX12 fallback in wgpu) refuses
1527
+ // to unroll a loop that contains an implicit-gradient texture sample when the
1528
+ // iteration count is uniform-driven, and refuses to *not* unroll because the
1529
+ // body has the gradient op — the only escape is to take the gradient out of
1530
+ // the loop. textureSampleLevel forces explicit LOD and removes the gradient
1531
+ // op, which is also what we want here (depth has no mips).
1532
+ for (var i = 0u; i < n_steps; i = i + 1u) {
1533
+ let ray_view = view_pos + r * t;
1534
+ let ray_clip = u.proj * vec4<f32>(ray_view, 1.0);
1535
+ let ray_ndc = ray_clip.xyz / ray_clip.w;
1536
+ if (ray_ndc.x < -1.0 || ray_ndc.x > 1.0 ||
1537
+ ray_ndc.y < -1.0 || ray_ndc.y > 1.0 ||
1538
+ ray_ndc.z < 0.0 || ray_ndc.z > 1.0) {
1539
+ break;
1540
+ }
1541
+ let ray_uv = vec2<f32>(ray_ndc.x * 0.5 + 0.5, 1.0 - (ray_ndc.y * 0.5 + 0.5));
1542
+ let scene_depth = textureSampleLevel(depth_tex, depth_samp, ray_uv, 0i);
1543
+
1544
+ if (ray_ndc.z >= scene_depth) {
1545
+ let hit_view = view_pos_from_depth(ray_uv, scene_depth);
1546
+ let thickness = abs(ray_view.z - hit_view.z);
1547
+ let step_world = t - prev_t;
1548
+ if (thickness < step_world * 2.0 + 0.1) {
1549
+ hit_uv = ray_uv;
1550
+ hit_found = true;
1551
+ }
1552
+ break;
1553
+ }
1554
+ prev_t = t;
1555
+ t = t + step_size;
1556
+ }
1557
+ if (!hit_found) {
1558
+ // March left the screen or found nothing — env fallback (EN-021).
1559
+ let fb = env_fallback(r, roughness) * fresnel * roughness_fade * u.params.x;
1560
+ let fb_safe = select(vec3<f32>(0.0), fb, fb == fb);
1561
+ return vec4<f32>(fb_safe, 0.0);
1562
+ }
1563
+
1564
+ let edge_fade = min(
1565
+ min(hit_uv.x, 1.0 - hit_uv.x),
1566
+ min(hit_uv.y, 1.0 - hit_uv.y),
1567
+ ) * 10.0;
1568
+ let fade = clamp(edge_fade, 0.0, 1.0);
1569
+
1570
+ // NaN scrubber: WGSL has no isnan(), but NaN == NaN is false for
1571
+ // every compliant backend, so a componentwise self-compare gives us
1572
+ // a vec3<bool> that is true iff each channel is finite. This nukes
1573
+ // the one stray NaN/Inf pixel per few thousand rays that would
1574
+ // otherwise ping-pong through ssr_history and tonemap to pink. Same
1575
+ // self-compare is applied to the HDR tap in case upstream writes a
1576
+ // bad sample (autoexposure ratios, rare shader ops on degenerate
1577
+ // triangles, etc).
1578
+ let raw = textureSampleLevel(hdr_tex, hdr_samp, hit_uv, 0.0).rgb;
1579
+ let reflected = select(vec3<f32>(0.0), raw, raw == raw);
1580
+ let out = reflected * fresnel * roughness_fade * u.params.x * fade;
1581
+ let out_safe = select(vec3<f32>(0.0), out, out == out);
1582
+ return vec4<f32>(out_safe, fade);
1583
+ }
1584
+ ";
1585
+
1586
+