@bornengine/engine 0.4.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (213) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +231 -0
  3. package/native/android/Cargo.lock +1848 -0
  4. package/native/android/Cargo.toml +24 -0
  5. package/native/android/src/lib.rs +702 -0
  6. package/native/ios/Cargo.lock +1690 -0
  7. package/native/ios/Cargo.toml +32 -0
  8. package/native/ios/src/lib.rs +1267 -0
  9. package/native/linux/Cargo.lock +3279 -0
  10. package/native/linux/Cargo.toml +29 -0
  11. package/native/linux/src/lib.rs +1331 -0
  12. package/native/macos/Cargo.lock +3310 -0
  13. package/native/macos/Cargo.toml +46 -0
  14. package/native/macos/src/lib.rs +1302 -0
  15. package/native/shared/Cargo.lock +1899 -0
  16. package/native/shared/Cargo.toml +62 -0
  17. package/native/shared/assets/default_font.ttf +0 -0
  18. package/native/shared/build.rs +270 -0
  19. package/native/shared/shaders/common/clouds.wgsl +122 -0
  20. package/native/shared/shaders/common/fog.wgsl +16 -0
  21. package/native/shared/shaders/common/foliage_wind.wgsl +98 -0
  22. package/native/shared/shaders/common/imposter.wgsl +112 -0
  23. package/native/shared/shaders/common/pbr.wgsl +186 -0
  24. package/native/shared/shaders/common/shadows.wgsl +186 -0
  25. package/native/shared/shaders/common/sky.wgsl +8 -0
  26. package/native/shared/shaders/common/tonemap.wgsl +25 -0
  27. package/native/shared/shaders/impulse_field.wgsl +57 -0
  28. package/native/shared/shaders/material_abi.wgsl +383 -0
  29. package/native/shared/shaders/materials/test_minimal.wgsl +42 -0
  30. package/native/shared/src/anim_mixer.rs +61 -0
  31. package/native/shared/src/attach.rs +263 -0
  32. package/native/shared/src/audio/decode.rs +123 -0
  33. package/native/shared/src/audio/mod.rs +863 -0
  34. package/native/shared/src/audio/render.rs +892 -0
  35. package/native/shared/src/audio/spsc.rs +156 -0
  36. package/native/shared/src/audio/stream.rs +226 -0
  37. package/native/shared/src/custom_shaders.rs +104 -0
  38. package/native/shared/src/decals.rs +245 -0
  39. package/native/shared/src/drs.rs +211 -0
  40. package/native/shared/src/engine.rs +261 -0
  41. package/native/shared/src/ffi.rs +116 -0
  42. package/native/shared/src/ffi_core/assets.rs +388 -0
  43. package/native/shared/src/ffi_core/audio_ffi.rs +184 -0
  44. package/native/shared/src/ffi_core/draw.rs +334 -0
  45. package/native/shared/src/ffi_core/game_loop.rs +577 -0
  46. package/native/shared/src/ffi_core/input.rs +234 -0
  47. package/native/shared/src/ffi_core/mod.rs +127 -0
  48. package/native/shared/src/ffi_core/models.rs +1154 -0
  49. package/native/shared/src/ffi_core/ragdoll_ffi.rs +261 -0
  50. package/native/shared/src/ffi_core/scene.rs +626 -0
  51. package/native/shared/src/ffi_core/vfx.rs +212 -0
  52. package/native/shared/src/ffi_core/visual.rs +691 -0
  53. package/native/shared/src/frame_callbacks.rs +122 -0
  54. package/native/shared/src/geometry.rs +236 -0
  55. package/native/shared/src/handles.rs +182 -0
  56. package/native/shared/src/input.rs +448 -0
  57. package/native/shared/src/jolt_sys.rs +822 -0
  58. package/native/shared/src/lib.rs +55 -0
  59. package/native/shared/src/models.rs +1093 -0
  60. package/native/shared/src/models_gltf.rs +1280 -0
  61. package/native/shared/src/particles.rs +391 -0
  62. package/native/shared/src/physics_jolt.rs +1908 -0
  63. package/native/shared/src/picking.rs +298 -0
  64. package/native/shared/src/postfx.rs +345 -0
  65. package/native/shared/src/profiler.rs +492 -0
  66. package/native/shared/src/ragdoll.rs +474 -0
  67. package/native/shared/src/renderer/atmosphere_lut.rs +573 -0
  68. package/native/shared/src/renderer/brdf_lut.rs +154 -0
  69. package/native/shared/src/renderer/draw2d.rs +143 -0
  70. package/native/shared/src/renderer/formats.rs +822 -0
  71. package/native/shared/src/renderer/froxel.rs +421 -0
  72. package/native/shared/src/renderer/gi_bake.rs +653 -0
  73. package/native/shared/src/renderer/graph.rs +462 -0
  74. package/native/shared/src/renderer/hiz.rs +269 -0
  75. package/native/shared/src/renderer/hot_reload.rs +390 -0
  76. package/native/shared/src/renderer/impulse_field.rs +456 -0
  77. package/native/shared/src/renderer/lighting.rs +154 -0
  78. package/native/shared/src/renderer/material_instancing.rs +171 -0
  79. package/native/shared/src/renderer/material_pipeline.rs +700 -0
  80. package/native/shared/src/renderer/material_system.rs +1996 -0
  81. package/native/shared/src/renderer/material_system_tests.rs +601 -0
  82. package/native/shared/src/renderer/material_system_wasm.rs +41 -0
  83. package/native/shared/src/renderer/mod.rs +12556 -0
  84. package/native/shared/src/renderer/model_draw.rs +641 -0
  85. package/native/shared/src/renderer/occlusion.rs +429 -0
  86. package/native/shared/src/renderer/planar_pass.rs +593 -0
  87. package/native/shared/src/renderer/planar_reflection.rs +499 -0
  88. package/native/shared/src/renderer/post_pass.rs +249 -0
  89. package/native/shared/src/renderer/postfx_chain.rs +728 -0
  90. package/native/shared/src/renderer/pt_pass.rs +577 -0
  91. package/native/shared/src/renderer/scene_pass.rs +607 -0
  92. package/native/shared/src/renderer/shader_include.rs +205 -0
  93. package/native/shared/src/renderer/shader_library.rs +135 -0
  94. package/native/shared/src/renderer/shaders/ao.rs +570 -0
  95. package/native/shared/src/renderer/shaders/core.rs +1243 -0
  96. package/native/shared/src/renderer/shaders/env.rs +907 -0
  97. package/native/shared/src/renderer/shaders/gi.rs +810 -0
  98. package/native/shared/src/renderer/shaders/mod.rs +19 -0
  99. package/native/shared/src/renderer/shaders/post.rs +1558 -0
  100. package/native/shared/src/renderer/shaders/pt.rs +1859 -0
  101. package/native/shared/src/renderer/shaders/ssgi.rs +1586 -0
  102. package/native/shared/src/renderer/shadow_pass.rs +731 -0
  103. package/native/shared/src/renderer/ssgi_pass.rs +392 -0
  104. package/native/shared/src/renderer/ssr_pass.rs +188 -0
  105. package/native/shared/src/renderer/texture_store.rs +473 -0
  106. package/native/shared/src/renderer/transient.rs +591 -0
  107. package/native/shared/src/renderer/types.rs +941 -0
  108. package/native/shared/src/renderer/util.rs +152 -0
  109. package/native/shared/src/scene.rs +1362 -0
  110. package/native/shared/src/sdf_cache.rs +274 -0
  111. package/native/shared/src/shadows.rs +1036 -0
  112. package/native/shared/src/staging.rs +102 -0
  113. package/native/shared/src/string_header.rs +266 -0
  114. package/native/shared/src/text_renderer.rs +502 -0
  115. package/native/shared/src/textures.rs +197 -0
  116. package/native/tvos/Cargo.lock +1693 -0
  117. package/native/tvos/Cargo.toml +36 -0
  118. package/native/tvos/metal-patched/Cargo.toml +178 -0
  119. package/native/tvos/metal-patched/LICENSE-APACHE +201 -0
  120. package/native/tvos/metal-patched/LICENSE-MIT +25 -0
  121. package/native/tvos/metal-patched/src/acceleration_structure.rs +667 -0
  122. package/native/tvos/metal-patched/src/acceleration_structure_pass.rs +108 -0
  123. package/native/tvos/metal-patched/src/argument.rs +366 -0
  124. package/native/tvos/metal-patched/src/blitpass.rs +102 -0
  125. package/native/tvos/metal-patched/src/buffer.rs +71 -0
  126. package/native/tvos/metal-patched/src/capturedescriptor.rs +76 -0
  127. package/native/tvos/metal-patched/src/capturemanager.rs +113 -0
  128. package/native/tvos/metal-patched/src/commandbuffer.rs +192 -0
  129. package/native/tvos/metal-patched/src/commandqueue.rs +44 -0
  130. package/native/tvos/metal-patched/src/computepass.rs +107 -0
  131. package/native/tvos/metal-patched/src/constants.rs +152 -0
  132. package/native/tvos/metal-patched/src/counters.rs +119 -0
  133. package/native/tvos/metal-patched/src/depthstencil.rs +190 -0
  134. package/native/tvos/metal-patched/src/device.rs +2134 -0
  135. package/native/tvos/metal-patched/src/drawable.rs +39 -0
  136. package/native/tvos/metal-patched/src/encoder.rs +2041 -0
  137. package/native/tvos/metal-patched/src/heap.rs +281 -0
  138. package/native/tvos/metal-patched/src/indirect_encoder.rs +344 -0
  139. package/native/tvos/metal-patched/src/lib.rs +657 -0
  140. package/native/tvos/metal-patched/src/library.rs +902 -0
  141. package/native/tvos/metal-patched/src/mps.rs +575 -0
  142. package/native/tvos/metal-patched/src/pipeline/compute.rs +475 -0
  143. package/native/tvos/metal-patched/src/pipeline/mod.rs +71 -0
  144. package/native/tvos/metal-patched/src/pipeline/render.rs +762 -0
  145. package/native/tvos/metal-patched/src/renderpass.rs +443 -0
  146. package/native/tvos/metal-patched/src/resource.rs +182 -0
  147. package/native/tvos/metal-patched/src/sampler.rs +165 -0
  148. package/native/tvos/metal-patched/src/sync.rs +178 -0
  149. package/native/tvos/metal-patched/src/texture.rs +352 -0
  150. package/native/tvos/metal-patched/src/types.rs +90 -0
  151. package/native/tvos/metal-patched/src/vertexdescriptor.rs +250 -0
  152. package/native/tvos/src/audio_backend.rs +197 -0
  153. package/native/tvos/src/lib.rs +1891 -0
  154. package/native/visionos/Cargo.lock +1693 -0
  155. package/native/visionos/Cargo.toml +40 -0
  156. package/native/visionos/src/audio_backend.rs +197 -0
  157. package/native/visionos/src/lib.rs +1887 -0
  158. package/native/watchos/Cargo.lock +16 -0
  159. package/native/watchos/Cargo.toml +19 -0
  160. package/native/watchos/shaders/bloom_postfx.metal +99 -0
  161. package/native/watchos/src/BloomWatchApp.swift +1267 -0
  162. package/native/watchos/src/BloomWatchAudio.swift +179 -0
  163. package/native/watchos/src/audio.rs +55 -0
  164. package/native/watchos/src/draw_list.rs +229 -0
  165. package/native/watchos/src/ffi_stubs.rs +915 -0
  166. package/native/watchos/src/ffi_stubs_manual.rs +35 -0
  167. package/native/watchos/src/lib.rs +1124 -0
  168. package/native/watchos/src/models.rs +746 -0
  169. package/native/watchos/src/postfx.rs +95 -0
  170. package/native/watchos/src/scene.rs +534 -0
  171. package/native/watchos/src/textures.rs +184 -0
  172. package/native/web/Cargo.lock +1657 -0
  173. package/native/web/Cargo.toml +43 -0
  174. package/native/web/bloom_glue.js +695 -0
  175. package/native/web/build.sh +131 -0
  176. package/native/web/index.html +35 -0
  177. package/native/web/jolt_bridge.js +1519 -0
  178. package/native/web/src/input_ffi.rs +286 -0
  179. package/native/web/src/lib.rs +1796 -0
  180. package/native/web/src/material_ffi.rs +710 -0
  181. package/native/web/src/parity_ffi.rs +343 -0
  182. package/native/web/src/physics_ffi.rs +643 -0
  183. package/native/web/src/ragdoll_ffi.rs +250 -0
  184. package/native/web/src/render_settings.rs +98 -0
  185. package/native/windows/Cargo.lock +1815 -0
  186. package/native/windows/Cargo.toml +68 -0
  187. package/native/windows/src/lib.rs +1486 -0
  188. package/package.json +4279 -0
  189. package/src/audio/index.ts +315 -0
  190. package/src/core/colors.ts +63 -0
  191. package/src/core/index.ts +1206 -0
  192. package/src/core/keys.ts +63 -0
  193. package/src/core/types.ts +104 -0
  194. package/src/index.ts +171 -0
  195. package/src/math/index.ts +516 -0
  196. package/src/mobile/index.ts +294 -0
  197. package/src/models/index.ts +1258 -0
  198. package/src/physics/index.ts +1134 -0
  199. package/src/scene/index.ts +698 -0
  200. package/src/shapes/index.ts +120 -0
  201. package/src/text/index.ts +48 -0
  202. package/src/textures/index.ts +187 -0
  203. package/src/vfx/index.ts +191 -0
  204. package/src/world/index.ts +24 -0
  205. package/src/world/loader.ts +423 -0
  206. package/src/world/prefab.ts +217 -0
  207. package/src/world/render.ts +172 -0
  208. package/src/world/saver.ts +108 -0
  209. package/src/world/serialize.ts +301 -0
  210. package/src/world/terrain.ts +355 -0
  211. package/src/world/types.ts +160 -0
  212. package/src/world/validate.ts +319 -0
  213. package/src/world/version.ts +114 -0
@@ -0,0 +1,1859 @@
1
+ //! Path-tracing megakernel (docs/pt/pt-roadmap.md, ticket PT-1).
2
+ //!
3
+ //! One compute kernel, one ray budget per pixel per frame. Primary hits come
4
+ //! from the G-buffer (depth + albedo + material MRTs — free, and sharper than
5
+ //! traced primaries); bounce and shadow rays go through the same TLAS the
6
+ //! Lumen HW probe trace uses. Hit shading at bounces reads the mesh-card
7
+ //! ALBEDO atlas (not the pre-lit radiance atlas: a path tracer computes its
8
+ //! own lighting at every vertex of the path — sampling pre-lit cards would
9
+ //! bake Lumen's direct light into ours twice).
10
+ //!
11
+ //! Radiometric convention: light intensities are treated as π-premultiplied,
12
+ //! i.e. diffuse contribution is `albedo * L * NdotL` with no 1/π — matching
13
+ //! the raster shader (core.rs point-light loop has no 1/π either), so
14
+ //! toggling PT on/off does not jump scene brightness. bloom-reference
15
+ //! comparisons account for this in scene config (see the PT-1 ticket).
16
+ //!
17
+ //! Sky pixels are never written: the raster sky/cloud passes already drew
18
+ //! them, and PT replacing a procedural cloud deck with an analytic gradient
19
+ //! would be a downgrade. PT owns geometry pixels only. The translucent pass
20
+ //! runs AFTER this kernel, so water and glass composite over path-traced
21
+ //! opaques exactly as they do over raster ones.
22
+ //!
23
+ //! Debug modes (uniform cfg.w, set via BLOOM_PT_DEBUG):
24
+ //! 1 = raw depth visualised 2 = reconstructed world normals
25
+ //! 3 = G-buffer albedo 4 = sun shadow-ray visibility
26
+ //! 5 = solid magenta (pipeline probe — proves dispatch + write path)
27
+ //! 6 = traced-primary interpolated normal (compare against 2;
28
+ //! magenta = TLAS miss where G-buffer had geometry, orange = hit
29
+ //! instance without a geometry window)
30
+ //! 7 = traced-primary textured hit albedo (compare against 3;
31
+ //! yellow = adapter lacks texture-array features)
32
+ //! 8-15 = binary/quantized probes from the DX12 bring-up (hit-window
33
+ //! flag, normal axes, primitive/instance banding, two-query
34
+ //! aliasing, t-vs-G-buffer sanity, t contours). 13 is the
35
+ //! keeper: green = traced t agrees with the G-buffer, red =
36
+ //! mismatch, blue = miss.
37
+ //! 16/17 = NUMERIC dumps via the accum buffer + CPU readback
38
+ //! (pt_trace_dump.txt): 16 = t/instance/prim/kind, 17 = p0 +
39
+ //! raw depth. These found the transposed inv_vp: when every
40
+ //! probe looks "constant", dump numbers before theorizing.
41
+
42
+ pub(in crate::renderer) const PT_KERNEL_WGSL: &str = r#"
43
+ struct PtLight {
44
+ pos_range: vec4<f32>, // xyz world position, w = range
45
+ color_int: vec4<f32>, // rgb color, w = intensity
46
+ };
47
+
48
+ struct PtParams {
49
+ inv_vp: mat4x4<f32>,
50
+ // PT-3: previous frame's UNJITTERED view-projection — reprojects
51
+ // this frame's world positions into last frame's screen for
52
+ // temporal history fetch in realtime mode.
53
+ prev_vp: mat4x4<f32>,
54
+ cam_pos: vec4<f32>, // xyz camera world pos
55
+ sun_dir: vec4<f32>, // xyz unit vector toward the sun
56
+ sun_color: vec4<f32>, // rgb premultiplied by intensity
57
+ sky_color: vec4<f32>, // rgb ambient-derived sky tint
58
+ size: vec4<u32>, // x/y = TRACE grid dims, z=frame_index, w=accum_count
59
+ cfg: vec4<f32>, // x=mode(1|2), y=max_bounces, z=point_light_count, w=debug
60
+ // PT-3 half-res: x/y = full G-buffer dims. z = 1 -> hybrid sun
61
+ // (sample the raster shadow cascades instead of tracing the sun;
62
+ // crisp noise-free direct shadows, rays spent on indirect only).
63
+ // w = 1 -> ReSTIR DI (PT-4, experimental).
64
+ ext: vec4<u32>,
65
+ // Raster shadow cascade view-projections. Uploaded RAW, like
66
+ // prev_vp: mat4_multiply products are already in WGSL M*v layout;
67
+ // only mat4_invert outputs (inv_vp) upload transposed.
68
+ shadow_vps: array<mat4x4<f32>, 3>,
69
+ lights: array<PtLight, 16>,
70
+ };
71
+
72
+ // Layout mirror of the Lumen instance data (ssgi.rs) — same buffer.
73
+ struct InstanceGiData {
74
+ albedo: vec3<f32>,
75
+ emissive_luma: f32,
76
+ normal_ws: vec3<f32>,
77
+ _pad0: f32,
78
+ card_slot: vec4<f32>,
79
+ card_aabb_min: vec4<f32>,
80
+ card_aabb_max: vec4<f32>,
81
+ world_aabb_min: vec4<f32>,
82
+ world_aabb_max: vec4<f32>,
83
+ // PT-2: x = vertex_base, y = index_base, z = index_count (0 = no
84
+ // geometry window -> PT-1 fallback), w = albedo texture index.
85
+ geo: vec4<u32>,
86
+ // PT-2: x = roughness, y = metalness.
87
+ mat_params: vec4<f32>,
88
+ };
89
+
90
+ @group(0) @binding(0) var<uniform> u: PtParams;
91
+ @group(0) @binding(1) var accel: acceleration_structure;
92
+ @group(0) @binding(2) var<storage, read> instance_data: array<InstanceGiData>;
93
+ @group(0) @binding(3) var depth_tex: texture_depth_2d;
94
+ @group(0) @binding(4) var albedo_tex: texture_2d<f32>;
95
+ @group(0) @binding(5) var material_tex: texture_2d<f32>;
96
+ @group(0) @binding(6) var card_albedo_atlas: texture_2d<f32>;
97
+ @group(0) @binding(7) var card_samp: sampler;
98
+ // PT-3: ping-pong accumulation. Binding 8 = previous frame's buffer
99
+ // (read), binding 13 = this frame's output. Reprojection reads OTHER
100
+ // pixels from prev, which a single read_write buffer cannot do safely.
101
+ //
102
+ // Layout (SVGF): accum = (irradiance rgb, luminance variance);
103
+ // moments = (mu1, mu2, history length, raw depth). Progressive mode
104
+ // keeps its original (radiance sum, sample count) layout in accum and
105
+ // leaves the moments buffers untouched.
106
+ @group(0) @binding(8) var<storage, read_write> accum: array<vec4<f32>>;
107
+ @group(0) @binding(9) var out_hdr: texture_storage_2d<rgba16float, write>;
108
+ @group(0) @binding(13) var<storage, read_write> accum_out: array<vec4<f32>>;
109
+ @group(0) @binding(18) var<storage, read_write> moments: array<vec4<f32>>;
110
+ @group(0) @binding(19) var<storage, read_write> moments_out: array<vec4<f32>>;
111
+ // PT-4 (EXPERIMENTAL, ext.w == 1) — ReSTIR DI reservoirs, ping-pong
112
+ // with the accum pair: (light index, W, M, target pdf) per trace texel.
113
+ @group(0) @binding(20) var<storage, read_write> resv: array<vec4<f32>>;
114
+ @group(0) @binding(21) var<storage, read_write> resv_out: array<vec4<f32>>;
115
+ // PT-7 — the raster velocity MRT (uv-space delta, current − previous,
116
+ // no Y flip at write; see core.rs). Non-zero where a surface MOVED —
117
+ // reprojection follows it instead of the camera-only prev_vp math,
118
+ // exactly like TAA, so moving skinned characters keep their history.
119
+ @group(0) @binding(22) var velocity_tex: texture_2d<f32>;
120
+
121
+ // Reprojection of the current surface into the previous frame's trace
122
+ // grid — computed once per texel (module privates because both the
123
+ // ReSTIR temporal reuse and the SVGF colour accumulation consume it).
124
+ var<private> rp_valid: bool;
125
+ var<private> rp_base: vec2<i32>;
126
+ var<private> rp_fr: vec2<f32>;
127
+ var<private> rp_zl_here: f32;
128
+ var<private> rp_nearest: u32;
129
+
130
+ fn compute_reproj(p0: vec3<f32>, px_full: vec2<i32>, depth_cur: f32) {
131
+ rp_valid = false;
132
+ if (u.size.w == 0u) { return; }
133
+ // PT-7 — object motion first: the velocity buffer knows how THIS
134
+ // pixel's surface moved (including skeletal motion, which no
135
+ // camera matrix can express). TAA's convention:
136
+ // prev_uv = (uv.x - vel.x, uv.y + vel.y). Camera-only pixels
137
+ // write ~zero velocity and fall through to the prev_vp math.
138
+ var uv_prev: vec2<f32>;
139
+ var zl_here: f32;
140
+ let vel = textureLoad(velocity_tex, px_full, 0).rg;
141
+ if (abs(vel.x) + abs(vel.y) > 1e-5) {
142
+ let uv_cur = (vec2<f32>(px_full) + 0.5)
143
+ / vec2<f32>(f32(u.ext.x), f32(u.ext.y));
144
+ uv_prev = vec2<f32>(uv_cur.x - vel.x, uv_cur.y + vel.y);
145
+ // Depth along a moving surface changes slowly frame-to-frame;
146
+ // the tap tolerance absorbs it, and a fast approach degrades
147
+ // to a disocclusion reset — the safe direction.
148
+ zl_here = lin_depth(depth_cur);
149
+ } else {
150
+ let clip_prev = u.prev_vp * vec4<f32>(p0, 1.0);
151
+ if (clip_prev.w <= 1e-4) { return; }
152
+ let ndc_prev = clip_prev.xyz / clip_prev.w;
153
+ uv_prev = vec2<f32>(ndc_prev.x * 0.5 + 0.5, 0.5 - ndc_prev.y * 0.5);
154
+ zl_here = lin_depth(ndc_prev.z);
155
+ }
156
+ if (uv_prev.x < 0.0 || uv_prev.x >= 1.0 || uv_prev.y < 0.0 || uv_prev.y >= 1.0) {
157
+ return;
158
+ }
159
+ let pos = uv_prev * vec2<f32>(f32(u.size.x), f32(u.size.y)) - 0.5;
160
+ rp_base = vec2<i32>(floor(pos));
161
+ rp_fr = pos - floor(pos);
162
+ rp_zl_here = zl_here;
163
+ let np = vec2<u32>(
164
+ min(u32(max(rp_base.x + i32(round(rp_fr.x)), 0)), u.size.x - 1u),
165
+ min(u32(max(rp_base.y + i32(round(rp_fr.y)), 0)), u.size.y - 1u),
166
+ );
167
+ rp_nearest = np.y * u.size.x + np.x;
168
+ rp_valid = true;
169
+ }
170
+
171
+ // Approximate linear view distance from the raw depth-buffer value
172
+ // (GL-convention matrix, near 0.01: z_view ~= 2n / (1 - d)). Only used
173
+ // for RELATIVE history-validation comparisons.
174
+ fn lin_depth(d: f32) -> f32 {
175
+ return 0.02 / max(1.0 - d, 1e-6);
176
+ }
177
+ // PT-2: geometry megabuffers. geo_v holds raw Vertex3D words (stride 24
178
+ // f32: position +0, normal +3, color +6, uv +10, ...); geo_i holds the
179
+ // concatenated index streams. Windows are per-instance via inst.geo.
180
+ // (Binding 12, the texture array + PT_HAS_TEXTURES + pt_tex_sample, is
181
+ // appended by the Rust side per adapter support.)
182
+ @group(0) @binding(10) var<storage, read> geo_v: array<f32>;
183
+ @group(0) @binding(11) var<storage, read> geo_i: array<u32>;
184
+ // Hybrid sun (ext.z == 1): the raster shadow cascades.
185
+ @group(0) @binding(14) var shadow_atlas_0: texture_depth_2d;
186
+ @group(0) @binding(15) var shadow_atlas_1: texture_depth_2d;
187
+ @group(0) @binding(16) var shadow_atlas_2: texture_depth_2d;
188
+ @group(0) @binding(17) var shadow_samp: sampler_comparison;
189
+
190
+ // Sun visibility from the shadow cascades (near -> far fallthrough by
191
+ // coverage, same scheme as the WSRC bake). Deterministic and smooth —
192
+ // the whole reason RT mode's direct light doesn't shimmer or dither.
193
+ fn sun_vis_cascade(pos_ws: vec3<f32>) -> f32 {
194
+ for (var c = 0; c < 3; c = c + 1) {
195
+ var clip: vec4<f32>;
196
+ if (c == 0) { clip = u.shadow_vps[0] * vec4<f32>(pos_ws, 1.0); }
197
+ else if (c == 1) { clip = u.shadow_vps[1] * vec4<f32>(pos_ws, 1.0); }
198
+ else { clip = u.shadow_vps[2] * vec4<f32>(pos_ws, 1.0); }
199
+ if (abs(clip.w) < 1e-6) { continue; }
200
+ let ndc = clip.xyz / clip.w;
201
+ if (ndc.x < -0.99 || ndc.x > 0.99 || ndc.y < -0.99 || ndc.y > 0.99 || ndc.z < 0.0 || ndc.z > 1.0) {
202
+ continue;
203
+ }
204
+ let uv = vec2<f32>(ndc.x * 0.5 + 0.5, 0.5 - ndc.y * 0.5);
205
+ let ref_depth = ndc.z - 0.002;
206
+ // Manual load-and-compare with a 2x2 average instead of the
207
+ // comparison sampler: SampleCmp from a COMPUTE stage proved
208
+ // unreliable on this DXC path (constant 0, independent of the
209
+ // matrices — same failure shape as the ray-query saga), and the
210
+ // only other compute-stage user (WSRC bake) was never validated
211
+ // on DX12. textureLoad is proven (the PT depth reads use it).
212
+ var dims: vec2<u32>;
213
+ if (c == 0) { dims = textureDimensions(shadow_atlas_0); }
214
+ else if (c == 1) { dims = textureDimensions(shadow_atlas_1); }
215
+ else { dims = textureDimensions(shadow_atlas_2); }
216
+ let fdims = vec2<f32>(dims);
217
+ var vis = 0.0;
218
+ for (var ty = 0; ty <= 1; ty = ty + 1) {
219
+ for (var tx = 0; tx <= 1; tx = tx + 1) {
220
+ let tc = clamp(
221
+ vec2<i32>(uv * fdims - vec2<f32>(0.5)) + vec2<i32>(tx, ty),
222
+ vec2<i32>(0),
223
+ vec2<i32>(i32(dims.x) - 1, i32(dims.y) - 1),
224
+ );
225
+ var stored: f32;
226
+ if (c == 0) { stored = textureLoad(shadow_atlas_0, tc, 0); }
227
+ else if (c == 1) { stored = textureLoad(shadow_atlas_1, tc, 0); }
228
+ else { stored = textureLoad(shadow_atlas_2, tc, 0); }
229
+ if (ref_depth <= stored) { vis += 0.25; }
230
+ }
231
+ }
232
+ return vis;
233
+ }
234
+ return 1.0;
235
+ }
236
+
237
+ const PT_VSTRIDE: u32 = 24u;
238
+
239
+ struct HitAttrs {
240
+ normal_os: vec3<f32>,
241
+ uv: vec2<f32>,
242
+ };
243
+
244
+ fn vert_normal_os(slot: u32) -> vec3<f32> {
245
+ let o = slot * PT_VSTRIDE + 3u;
246
+ return vec3<f32>(geo_v[o], geo_v[o + 1u], geo_v[o + 2u]);
247
+ }
248
+
249
+ fn vert_uv(slot: u32) -> vec2<f32> {
250
+ let o = slot * PT_VSTRIDE + 10u;
251
+ return vec2<f32>(geo_v[o], geo_v[o + 1u]);
252
+ }
253
+
254
+ // Interpolate the hit triangle's vertex normal + UV. DXR/Vulkan
255
+ // barycentric convention: (u, v) weight vertices 1 and 2, w = 1-u-v
256
+ // weights vertex 0.
257
+ fn fetch_hit_attrs(geo: vec4<u32>, prim: u32, bary: vec2<f32>) -> HitAttrs {
258
+ let base = geo.y + prim * 3u;
259
+ let s0 = geo.x + geo_i[base];
260
+ let s1 = geo.x + geo_i[base + 1u];
261
+ let s2 = geo.x + geo_i[base + 2u];
262
+ let w = 1.0 - bary.x - bary.y;
263
+ var a: HitAttrs;
264
+ a.normal_os = w * vert_normal_os(s0) + bary.x * vert_normal_os(s1) + bary.y * vert_normal_os(s2);
265
+ a.uv = w * vert_uv(s0) + bary.x * vert_uv(s1) + bary.y * vert_uv(s2);
266
+ return a;
267
+ }
268
+
269
+ // Object-space normal -> world space: with M = object_to_world the
270
+ // correct transform is (M^-1)^T, and the ray query hands us M^-1 as
271
+ // world_to_object. `v * mat3` multiplies by the transpose in WGSL.
272
+ fn normal_to_world(n_os: vec3<f32>, w2o: mat4x3<f32>) -> vec3<f32> {
273
+ let lin = mat3x3<f32>(w2o[0], w2o[1], w2o[2]);
274
+ let n = n_os * lin;
275
+ let len = length(n);
276
+ if (len < 1e-8) { return vec3<f32>(0.0, 1.0, 0.0); }
277
+ return n / len;
278
+ }
279
+
280
+ // ---- RNG: PCG, one stream per (pixel, frame) --------------------------------
281
+
282
+ var<private> rng_state: u32;
283
+
284
+ fn rng_seed(px: vec2<u32>, frame: u32) {
285
+ var h = px.x * 374761393u + px.y * 668265263u + frame * 2654435761u;
286
+ h = (h ^ (h >> 13u)) * 1274126177u;
287
+ rng_state = h ^ (h >> 16u);
288
+ }
289
+
290
+ fn rand_f() -> f32 {
291
+ // PCG-XSH-RR step.
292
+ let old = rng_state;
293
+ rng_state = old * 747796405u + 2891336453u;
294
+ let word = ((old >> ((old >> 28u) + 4u)) ^ old) * 277803737u;
295
+ let out = (word >> 22u) ^ word;
296
+ return f32(out) * 2.3283064e-10; // / 2^32
297
+ }
298
+
299
+ fn rand_2f() -> vec2<f32> { return vec2<f32>(rand_f(), rand_f()); }
300
+
301
+ // Interleaved gradient noise, scrolled per frame by the golden-ratio
302
+ // offset. Spatially STRUCTURED (neighbors get well-distributed values)
303
+ // so a 5x5 filter averages it nearly flat — white PCG noise leaves
304
+ // mid-frequency blotch at 1-2 spp that shimmers. Used for the primary
305
+ // sun test in realtime mode.
306
+ fn ign_at(px: vec2<i32>, frame: u32) -> f32 {
307
+ let p = vec2<f32>(px) + f32(frame % 64u) * 5.588238;
308
+ return fract(52.9829189 * fract(0.06711056 * p.x + 0.00583715 * p.y));
309
+ }
310
+
311
+ // ---- Geometry reconstruction -------------------------------------------------
312
+
313
+ // px here is always a FULL-resolution G-buffer pixel (u.ext dims); the
314
+ // trace grid may be half of that in realtime mode.
315
+ fn world_at(px: vec2<i32>, depth: f32) -> vec3<f32> {
316
+ let dims = vec2<f32>(f32(u.ext.x), f32(u.ext.y));
317
+ let uv = (vec2<f32>(px) + vec2<f32>(0.5)) / dims;
318
+ let ndc = vec4<f32>(uv.x * 2.0 - 1.0, 1.0 - uv.y * 2.0, depth, 1.0);
319
+ let w = u.inv_vp * ndc;
320
+ return w.xyz / w.w;
321
+ }
322
+
323
+ fn depth_at(px: vec2<i32>) -> f32 {
324
+ let clamped = clamp(px, vec2<i32>(0), vec2<i32>(i32(u.ext.x) - 1, i32(u.ext.y) - 1));
325
+ return textureLoad(depth_tex, clamped, 0);
326
+ }
327
+
328
+ fn is_sky(depth: f32) -> bool {
329
+ // Depth-buffer far plane. If the projection turns out reversed-Z the
330
+ // BLOOM_PT_DEBUG=1 depth view makes it obvious in one screenshot; flip
331
+ // here if geometry reads bright and sky reads dark.
332
+ return depth >= 0.9999999;
333
+ }
334
+
335
+ // Screen-space normal from depth: reconstruct neighbours, take the tighter
336
+ // derivative on each axis so depth discontinuities don't smear normals
337
+ // across silhouettes.
338
+ fn normal_from_depth(px: vec2<i32>, p_center: vec3<f32>) -> vec3<f32> {
339
+ let d_l = depth_at(px + vec2<i32>(-1, 0));
340
+ let d_r = depth_at(px + vec2<i32>(1, 0));
341
+ let d_u = depth_at(px + vec2<i32>(0, -1));
342
+ let d_d = depth_at(px + vec2<i32>(0, 1));
343
+ let d_c = depth_at(px);
344
+
345
+ var ddx: vec3<f32>;
346
+ if (abs(d_l - d_c) < abs(d_r - d_c)) {
347
+ ddx = p_center - world_at(px + vec2<i32>(-1, 0), d_l);
348
+ } else {
349
+ ddx = world_at(px + vec2<i32>(1, 0), d_r) - p_center;
350
+ }
351
+ var ddy: vec3<f32>;
352
+ if (abs(d_u - d_c) < abs(d_d - d_c)) {
353
+ ddy = p_center - world_at(px + vec2<i32>(0, -1), d_u);
354
+ } else {
355
+ ddy = world_at(px + vec2<i32>(0, 1), d_d) - p_center;
356
+ }
357
+ var n = cross(ddy, ddx);
358
+ let len = length(n);
359
+ if (len < 1e-8) { return vec3<f32>(0.0, 1.0, 0.0); }
360
+ n = n / len;
361
+ // Face the camera: a G-buffer surface always does.
362
+ if (dot(n, u.cam_pos.xyz - p_center) < 0.0) { n = -n; }
363
+ return n;
364
+ }
365
+
366
+ // ---- Sampling helpers ----------------------------------------------------------
367
+
368
+ // Branchless ONB (Duff et al. 2017).
369
+ fn onb(n: vec3<f32>) -> mat3x3<f32> {
370
+ let s = select(-1.0, 1.0, n.z >= 0.0);
371
+ let a = -1.0 / (s + n.z);
372
+ let b = n.x * n.y * a;
373
+ let t = vec3<f32>(1.0 + s * n.x * n.x * a, s * b, -s * n.x);
374
+ let bt = vec3<f32>(b, s + n.y * n.y * a, -n.y);
375
+ return mat3x3<f32>(t, bt, n);
376
+ }
377
+
378
+ fn cosine_sample(n: vec3<f32>, r: vec2<f32>) -> vec3<f32> {
379
+ let phi = 6.2831853 * r.x;
380
+ let sr = sqrt(r.y);
381
+ let local = vec3<f32>(cos(phi) * sr, sin(phi) * sr, sqrt(max(0.0, 1.0 - r.y)));
382
+ return normalize(onb(n) * local);
383
+ }
384
+
385
+ // Uniform direction in the solar cone (half-angle 0.265 deg -> soft shadows).
386
+ fn sun_cone_sample(r: vec2<f32>) -> vec3<f32> {
387
+ let cos_max = 0.9999893;
388
+ let cos_t = mix(cos_max, 1.0, r.x);
389
+ let sin_t = sqrt(max(0.0, 1.0 - cos_t * cos_t));
390
+ let phi = 6.2831853 * r.y;
391
+ let local = vec3<f32>(cos(phi) * sin_t, sin(phi) * sin_t, cos_t);
392
+ return normalize(onb(u.sun_dir.xyz) * local);
393
+ }
394
+
395
+ // Sky radiance for a miss. Analytic horizon-to-zenith gradient off the same
396
+ // ambient-derived tint Lumen's traces use; the sun disc is deliberately
397
+ // absent (the sun is sampled by NEE only, so it cannot be counted twice).
398
+ fn sky_radiance(dir: vec3<f32>) -> vec3<f32> {
399
+ let t = clamp(dir.y * 0.5 + 0.5, 0.0, 1.0);
400
+ return u.sky_color.rgb * mix(0.45, 1.35, t);
401
+ }
402
+
403
+ // ---- GGX BRDF sampling (PT-2; port of bloom-reference sample_brdf) --------
404
+
405
+ fn fresnel_schlick3(cos_theta: f32, f0: vec3<f32>) -> vec3<f32> {
406
+ let m = clamp(1.0 - cos_theta, 0.0, 1.0);
407
+ let m2 = m * m;
408
+ return f0 + (vec3<f32>(1.0) - f0) * (m2 * m2 * m);
409
+ }
410
+
411
+ fn smith_g1(n_dot_x: f32, alpha: f32) -> f32 {
412
+ let a2 = alpha * alpha;
413
+ let inner = sqrt((1.0 - a2) * n_dot_x * n_dot_x + a2);
414
+ return 2.0 * n_dot_x / (n_dot_x + inner + 1e-6);
415
+ }
416
+
417
+ fn v_smith(n_dot_v: f32, n_dot_l: f32, alpha: f32) -> f32 {
418
+ let a2 = alpha * alpha;
419
+ let ggx_v = n_dot_l * sqrt((n_dot_v * (1.0 - a2) + a2) * n_dot_v);
420
+ let ggx_l = n_dot_v * sqrt((n_dot_l * (1.0 - a2) + a2) * n_dot_l);
421
+ return 0.5 / (ggx_v + ggx_l + 1e-6);
422
+ }
423
+
424
+ fn burley_diffuse(n_dot_l: f32, n_dot_v: f32, l_dot_h: f32, roughness: f32) -> f32 {
425
+ let fd90 = 0.5 + 2.0 * l_dot_h * l_dot_h * roughness;
426
+ let ml = pow(1.0 - n_dot_l, 5.0);
427
+ let mv = pow(1.0 - n_dot_v, 5.0);
428
+ return (1.0 + (fd90 - 1.0) * ml) * (1.0 + (fd90 - 1.0) * mv) / 3.14159265;
429
+ }
430
+
431
+ // Heitz 2018 VNDF sampler — visible-normal distribution, tangent frame.
432
+ fn sample_ggx_vndf(v_t: vec3<f32>, alpha: f32, r2: vec2<f32>) -> vec3<f32> {
433
+ let vh = normalize(vec3<f32>(alpha * v_t.x, alpha * v_t.y, v_t.z));
434
+ let lensq = vh.x * vh.x + vh.y * vh.y;
435
+ var t1 = vec3<f32>(1.0, 0.0, 0.0);
436
+ if (lensq > 0.0) {
437
+ t1 = vec3<f32>(-vh.y, vh.x, 0.0) / sqrt(lensq);
438
+ }
439
+ let t2 = cross(vh, t1);
440
+ let r = sqrt(r2.x);
441
+ let phi = 6.2831853 * r2.y;
442
+ let t1v = r * cos(phi);
443
+ var t2v = r * sin(phi);
444
+ let s = 0.5 * (1.0 + vh.z);
445
+ t2v = (1.0 - s) * sqrt(max(0.0, 1.0 - t1v * t1v)) + s * t2v;
446
+ let nh = t1v * t1 + t2v * t2 + sqrt(max(0.0, 1.0 - t1v * t1v - t2v * t2v)) * vh;
447
+ return normalize(vec3<f32>(alpha * nh.x, alpha * nh.y, max(nh.z, 0.0)));
448
+ }
449
+
450
+ struct BrdfSample {
451
+ dir: vec3<f32>,
452
+ // BRDF * cos / pdf, physical convention. For the pure-diffuse case
453
+ // this reduces to plain albedo, so the game's pi-premultiplied
454
+ // light intensities are unaffected.
455
+ weight: vec3<f32>,
456
+ valid: bool,
457
+ };
458
+
459
+ fn sample_brdf(
460
+ n: vec3<f32>,
461
+ view_ws: vec3<f32>,
462
+ base_color: vec3<f32>,
463
+ roughness: f32,
464
+ metallic: f32,
465
+ ) -> BrdfSample {
466
+ var out: BrdfSample;
467
+ out.valid = false;
468
+ let alpha = max(roughness * roughness, 1e-3);
469
+ let m = onb(n); // columns (t, bt, n): local -> world
470
+ let v_t = vec3<f32>(dot(view_ws, m[0]), dot(view_ws, m[1]), dot(view_ws, n));
471
+ if (v_t.z <= 0.0) {
472
+ return out;
473
+ }
474
+ let f0 = mix(vec3<f32>(0.04), base_color, metallic);
475
+ // Lobe pick by Fresnel at the ACTUAL view angle, not at normal
476
+ // incidence: at grazing angles specular energy approaches 1, and
477
+ // the estimator divides by the pick probability — picking with the
478
+ // ~0.04 normal-incidence weight amplified rare grazing specular
479
+ // samples ~25x into a field of white fireflies at 1-2 spp (the
480
+ // whole ground plane is grazing at distance). Clamped so neither
481
+ // lobe's 1/p boost can exceed ~20x even in edge cases.
482
+ let n_dot_v_pick = max(dot(n, view_ws), 0.0);
483
+ let f_view = fresnel_schlick3(n_dot_v_pick, f0);
484
+ let spec_weight = (f_view.x + f_view.y + f_view.z) / 3.0;
485
+ let diff_weight = (1.0 - spec_weight) * (1.0 - metallic);
486
+ var p_spec = spec_weight / (spec_weight + diff_weight + 1e-6);
487
+ p_spec = clamp(p_spec, 0.05, 0.95);
488
+ let r2 = rand_2f();
489
+ if (rand_f() < p_spec) {
490
+ let h_t = sample_ggx_vndf(v_t, alpha, r2);
491
+ let l_t = reflect(-v_t, h_t);
492
+ if (l_t.z <= 0.0) {
493
+ return out;
494
+ }
495
+ let n_dot_l = l_t.z;
496
+ let n_dot_v = max(v_t.z, 1e-4);
497
+ let v_dot_h = max(dot(v_t, h_t), 1e-4);
498
+ let f = fresnel_schlick3(v_dot_h, f0);
499
+ // VNDF pdf: throughput collapses to F * G2 / G1(V).
500
+ let g2 = v_smith(n_dot_v, n_dot_l, alpha) * 4.0 * n_dot_v * n_dot_l;
501
+ let g1_v = smith_g1(n_dot_v, alpha);
502
+ out.dir = m * l_t;
503
+ out.weight = f * g2 / (max(g1_v, 1e-6) * p_spec);
504
+ // Realtime mode trades a little energy for stability: a single
505
+ // bounce may not multiply throughput more than 4x (the ~7-frame
506
+ // EMA window cannot average outliers away like progressive
507
+ // accumulation can). Progressive mode stays unclamped.
508
+ if (u.cfg.x >= 2.0) {
509
+ out.weight = min(out.weight, vec3<f32>(4.0));
510
+ }
511
+ out.valid = true;
512
+ return out;
513
+ }
514
+ // Diffuse lobe: cosine hemisphere; weight = albedo * burley * pi
515
+ // (Burley divides by pi internally; pdf = cos/pi cancels the cos).
516
+ let r = sqrt(r2.x);
517
+ let phi = 6.2831853 * r2.y;
518
+ let l_t = vec3<f32>(r * cos(phi), r * sin(phi), sqrt(max(0.0, 1.0 - r2.x)));
519
+ let n_dot_l = max(l_t.z, 1e-4);
520
+ let n_dot_v = max(v_t.z, 1e-4);
521
+ let h_un = v_t + l_t;
522
+ var l_dot_h = 0.0;
523
+ if (dot(h_un, h_un) > 1e-8) {
524
+ l_dot_h = max(dot(l_t, normalize(h_un)), 0.0);
525
+ }
526
+ let diffuse_albedo = base_color * (1.0 - metallic) * (vec3<f32>(1.0) - f0);
527
+ let fd = burley_diffuse(n_dot_l, n_dot_v, l_dot_h, roughness);
528
+ out.dir = m * l_t;
529
+ out.weight = diffuse_albedo * fd * 3.14159265 / (1.0 - p_spec);
530
+ if (u.cfg.x >= 2.0) {
531
+ out.weight = min(out.weight, vec3<f32>(4.0));
532
+ }
533
+ out.valid = true;
534
+ return out;
535
+ }
536
+
537
+ // ---- Ray casts ------------------------------------------------------------------
538
+
539
+ fn occluded(origin: vec3<f32>, dir: vec3<f32>, max_t: f32) -> bool {
540
+ var rq: ray_query;
541
+ rayQueryInitialize(&rq, accel, RayDesc(0u, 0xFFu, 0.001, max_t, origin, dir));
542
+ loop {
543
+ if (!rayQueryProceed(&rq)) { break; }
544
+ }
545
+ let hit = rayQueryGetCommittedIntersection(&rq);
546
+ return hit.kind != RAY_QUERY_INTERSECTION_NONE;
547
+ }
548
+
549
+ // ---- Hit shading: card albedo -----------------------------------------------------
550
+
551
+ // Same signed-axis card projection as the Lumen HW trace (ssgi.rs), but
552
+ // sampling the raw ALBEDO atlas. Falls back to the flat instance albedo when
553
+ // the mesh has no captured card.
554
+ fn albedo_at_hit(
555
+ inst: InstanceGiData,
556
+ hit_os: vec3<f32>,
557
+ dir_ws: vec3<f32>,
558
+ ) -> vec3<f32> {
559
+ if (inst.card_slot.w <= 0.5) {
560
+ return inst.albedo;
561
+ }
562
+ let abs_d = abs(dir_ws);
563
+ var axis_idx: u32 = 0u;
564
+ if (abs_d.y >= abs_d.x && abs_d.y >= abs_d.z) {
565
+ axis_idx = 2u;
566
+ } else if (abs_d.z >= abs_d.x) {
567
+ axis_idx = 4u;
568
+ }
569
+ var signed_axis: u32 = axis_idx;
570
+ if (axis_idx == 0u && dir_ws.x > 0.0) { signed_axis = 1u; }
571
+ else if (axis_idx == 2u && dir_ws.y > 0.0) { signed_axis = 3u; }
572
+ else if (axis_idx == 4u && dir_ws.z > 0.0) { signed_axis = 5u; }
573
+
574
+ let first_slot = u32(inst.card_slot.x);
575
+ let slot = first_slot + signed_axis;
576
+ let slot_x = slot % 64u;
577
+ let slot_y = slot / 64u;
578
+
579
+ let bmin = inst.card_aabb_min.xyz;
580
+ let bmax = inst.card_aabb_max.xyz;
581
+ var u_os: f32; var v_os: f32;
582
+ var u_lo: f32; var u_hi: f32; var v_lo: f32; var v_hi: f32;
583
+ var u_flip: f32 = 1.0;
584
+ if (signed_axis == 0u || signed_axis == 1u) {
585
+ u_os = hit_os.y; v_os = hit_os.z;
586
+ u_lo = bmin.y; u_hi = bmax.y; v_lo = bmin.z; v_hi = bmax.z;
587
+ if (signed_axis == 1u) { u_flip = -1.0; }
588
+ } else if (signed_axis == 2u || signed_axis == 3u) {
589
+ u_os = hit_os.x; v_os = hit_os.z;
590
+ u_lo = bmin.x; u_hi = bmax.x; v_lo = bmin.z; v_hi = bmax.z;
591
+ if (signed_axis == 3u) { u_flip = -1.0; }
592
+ } else {
593
+ u_os = hit_os.x; v_os = hit_os.y;
594
+ u_lo = bmin.x; u_hi = bmax.x; v_lo = bmin.y; v_hi = bmax.y;
595
+ if (signed_axis == 5u) { u_flip = -1.0; }
596
+ }
597
+ var u_norm = clamp((u_os - u_lo) / max(u_hi - u_lo, 1e-4), 0.0, 1.0);
598
+ let v_norm = clamp((v_os - v_lo) / max(v_hi - v_lo, 1e-4), 0.0, 1.0);
599
+ if (u_flip < 0.0) { u_norm = 1.0 - u_norm; }
600
+
601
+ let slot_size_uv = 1.0 / 64.0;
602
+ let texel_in_slot = slot_size_uv / 64.0;
603
+ let slot_u0 = f32(slot_x) * slot_size_uv + texel_in_slot;
604
+ let slot_v0 = f32(slot_y) * slot_size_uv + texel_in_slot;
605
+ let slot_span = slot_size_uv - 2.0 * texel_in_slot;
606
+ let atlas_uv = vec2<f32>(slot_u0 + u_norm * slot_span, slot_v0 + v_norm * slot_span);
607
+ return textureSampleLevel(card_albedo_atlas, card_samp, atlas_uv, 0.0).rgb;
608
+ }
609
+
610
+ // ---- Next-event estimation ---------------------------------------------------------
611
+
612
+ // Direct light at a surface point: sun through the solar cone + one point
613
+ // light chosen uniformly (contribution / pdf). Game-radiometry convention:
614
+ // no 1/pi (see file header).
615
+ // Sun visibility at a surface point: shadow cascades in hybrid mode
616
+ // (deterministic, matches the raster shadows exactly), a traced cone
617
+ // ray otherwise (reference quality, soft penumbra).
618
+ fn sun_visibility(p: vec3<f32>, n: vec3<f32>, r2: vec2<f32>) -> f32 {
619
+ if (u.ext.z == 1u) {
620
+ return sun_vis_cascade(p);
621
+ }
622
+ let sd = sun_cone_sample(r2);
623
+ if (dot(n, sd) <= 0.0) {
624
+ return 0.0;
625
+ }
626
+ if (occluded(p, sd, 1000.0)) {
627
+ return 0.0;
628
+ }
629
+ return 1.0;
630
+ }
631
+
632
+ // GGX highlight for an NEE light sample — the same D/F/V terms as
633
+ // sample_brdf, evaluated for a known light direction. The shadow ray is
634
+ // already paid for by the diffuse term, so specular NEE rides along
635
+ // free (PT-5: point lights and bounce vertices were diffuse-only, the
636
+ // documented PT-2 gap). Analytic lights cannot be hit by BSDF rays and
637
+ // sky misses exclude the sun disc, so nothing double-counts.
638
+ fn nee_spec(n: vec3<f32>, view: vec3<f32>, ldir: vec3<f32>, ndl: f32,
639
+ full_alb: vec3<f32>, rough: f32, metal: f32) -> vec3<f32> {
640
+ let hv = normalize(view + ldir);
641
+ let ndv = max(dot(n, view), 1e-4);
642
+ let ndh = max(dot(n, hv), 0.0);
643
+ let vdh = max(dot(view, hv), 1e-4);
644
+ let alpha0 = max(rough * rough, 1e-3);
645
+ let a2 = alpha0 * alpha0;
646
+ let dd = ndh * ndh * (a2 - 1.0) + 1.0;
647
+ let dterm = a2 / (3.14159265 * dd * dd);
648
+ let f0s = mix(vec3<f32>(0.04), full_alb, metal);
649
+ return fresnel_schlick3(vdh, f0s) * dterm * v_smith(ndv, ndl, alpha0) * ndl;
650
+ }
651
+
652
+ fn direct_light(p: vec3<f32>, n: vec3<f32>, alb: vec3<f32>, sun_r2: vec2<f32>,
653
+ view: vec3<f32>, full_alb: vec3<f32>, rough: f32, metal: f32,
654
+ with_points: bool) -> vec3<f32> {
655
+ var lit = vec3<f32>(0.0);
656
+ var spec = vec3<f32>(0.0);
657
+
658
+ let ndl = max(dot(n, u.sun_dir.xyz), 0.0);
659
+ if (ndl > 0.0) {
660
+ let vis = sun_visibility(p, n, sun_r2);
661
+ lit += u.sun_color.rgb * ndl * vis;
662
+ if (vis > 0.0) {
663
+ spec += nee_spec(n, view, u.sun_dir.xyz, ndl, full_alb, rough, metal)
664
+ * u.sun_color.rgb * vis;
665
+ }
666
+ }
667
+
668
+ let count = u32(u.cfg.z);
669
+ if (count > 0u && with_points) {
670
+ let pick = min(u32(rand_f() * f32(count)), count - 1u);
671
+ let l = u.lights[pick];
672
+ let to_l = l.pos_range.xyz - p;
673
+ let d = length(to_l);
674
+ let range = l.pos_range.w;
675
+ if (d < range && d > 1e-3) {
676
+ let dir = to_l / d;
677
+ let ndl2 = dot(n, dir);
678
+ if (ndl2 > 0.0 && !occluded(p, dir, d - 0.02)) {
679
+ // Raster-parity falloff: (1 - d/range)^2, core.rs.
680
+ let att = 1.0 - d / range;
681
+ let li = l.color_int.rgb * l.color_int.w * att * att * f32(count);
682
+ lit += li * ndl2;
683
+ spec += nee_spec(n, view, dir, ndl2, full_alb, rough, metal) * li;
684
+ }
685
+ }
686
+ }
687
+ return alb * lit + spec;
688
+ }
689
+
690
+ // ---- PT-4 (EXPERIMENTAL) — ReSTIR DI over the analytic point lights -------
691
+ //
692
+ // RIS with 8 uniform candidates + temporal reservoir reuse (M-capped at
693
+ // 20x), one shadow ray for the winner. The target is re-evaluated at
694
+ // the CURRENT shading point when merging history, so the temporal reuse
695
+ // carries no geometric bias; visibility reuse bias does not arise
696
+ // because visibility is never folded into the reservoir. With this
697
+ // game's <=16 analytic lights plain NEE is nearly as good (the roadmap
698
+ // said so up front) — this lands the architecture for the day emissive
699
+ // particles/muzzle flashes become real light sources.
700
+
701
+ // Unshadowed contribution of light `li` at the shading point.
702
+ fn restir_contrib(li: u32, p: vec3<f32>, n: vec3<f32>, view: vec3<f32>,
703
+ alb_diff: vec3<f32>, full_alb: vec3<f32>,
704
+ rough: f32, metal: f32) -> vec3<f32> {
705
+ let l = u.lights[li];
706
+ let to_l = l.pos_range.xyz - p;
707
+ let d = length(to_l);
708
+ let range = l.pos_range.w;
709
+ if (d >= range || d <= 1e-3) { return vec3<f32>(0.0); }
710
+ let dir = to_l / d;
711
+ let ndl = dot(n, dir);
712
+ if (ndl <= 0.0) { return vec3<f32>(0.0); }
713
+ let att = 1.0 - d / range;
714
+ let li_rgb = l.color_int.rgb * l.color_int.w * att * att;
715
+ return alb_diff * li_rgb * ndl
716
+ + nee_spec(n, view, dir, ndl, full_alb, rough, metal) * li_rgb;
717
+ }
718
+
719
+ // Scalar target density: luminance of the unshadowed contribution.
720
+ // Correctness never depends on the target — only variance does.
721
+ fn restir_target(li: u32, p: vec3<f32>, n: vec3<f32>, view: vec3<f32>,
722
+ alb_diff: vec3<f32>, full_alb: vec3<f32>,
723
+ rough: f32, metal: f32) -> f32 {
724
+ return dot(restir_contrib(li, p, n, view, alb_diff, full_alb, rough, metal),
725
+ vec3<f32>(0.2126, 0.7152, 0.0722));
726
+ }
727
+
728
+ // Runs at the primary vertex when ext.w == 1; writes this frame's
729
+ // reservoir and returns the winner's shadow-tested contribution.
730
+ fn restir_point_light(idx: u32, p: vec3<f32>, n: vec3<f32>, view: vec3<f32>,
731
+ alb_diff: vec3<f32>, full_alb: vec3<f32>,
732
+ rough: f32, metal: f32) -> vec3<f32> {
733
+ let count = u32(u.cfg.z);
734
+ if (count == 0u) {
735
+ resv_out[idx] = vec4<f32>(-1.0, 0.0, 0.0, 0.0);
736
+ return vec3<f32>(0.0);
737
+ }
738
+ var r_y = 0u;
739
+ var r_wsum = 0.0;
740
+ var r_m = 0.0;
741
+ var r_phat = 0.0;
742
+ // RIS: 8 uniform candidates (pdf = 1/count => w = phat * count).
743
+ for (var c = 0u; c < 8u; c = c + 1u) {
744
+ let cand = min(u32(rand_f() * f32(count)), count - 1u);
745
+ let ph = restir_target(cand, p, n, view, alb_diff, full_alb, rough, metal);
746
+ let w = ph * f32(count);
747
+ r_wsum += w;
748
+ r_m += 1.0;
749
+ if (w > 0.0 && rand_f() * r_wsum < w) {
750
+ r_y = cand;
751
+ r_phat = ph;
752
+ }
753
+ }
754
+ // Temporal reuse from the reprojected texel. The stored W already
755
+ // integrates that reservoir's history; its M is capped so stale
756
+ // samples cannot outvote fresh ones forever.
757
+ if (rp_valid) {
758
+ let pr = resv[rp_nearest];
759
+ let pm = min(pr.z, 160.0);
760
+ if (pm > 0.0 && pr.x >= 0.0 && u32(pr.x) < count) {
761
+ let py = u32(pr.x);
762
+ let ph = restir_target(py, p, n, view, alb_diff, full_alb, rough, metal);
763
+ let w = ph * pr.y * pm;
764
+ if (w > 0.0) {
765
+ r_wsum += w;
766
+ if (rand_f() * r_wsum < w) {
767
+ r_y = py;
768
+ r_phat = ph;
769
+ }
770
+ }
771
+ r_m += pm;
772
+ }
773
+ }
774
+ var r_w = 0.0;
775
+ if (r_phat > 0.0 && r_m > 0.0) {
776
+ r_w = r_wsum / (r_m * r_phat);
777
+ }
778
+ resv_out[idx] = vec4<f32>(f32(r_y), r_w, r_m, r_phat);
779
+ if (r_w <= 0.0) { return vec3<f32>(0.0); }
780
+ // One shadow ray for the winner.
781
+ let l = u.lights[r_y];
782
+ let to_l = l.pos_range.xyz - p;
783
+ let d = length(to_l);
784
+ if (d <= 1e-3) { return vec3<f32>(0.0); }
785
+ let dir = to_l / d;
786
+ if (occluded(p, dir, d - 0.02)) { return vec3<f32>(0.0); }
787
+ return restir_contrib(r_y, p, n, view, alb_diff, full_alb, rough, metal) * r_w;
788
+ }
789
+
790
+ // ---- Main -----------------------------------------------------------------------------
791
+
792
+ @compute @workgroup_size(8, 8, 1)
793
+ fn cs_main(@builtin(global_invocation_id) gid: vec3<u32>) {
794
+ if (gid.x >= u.size.x || gid.y >= u.size.y) { return; }
795
+ let px = vec2<i32>(i32(gid.x), i32(gid.y));
796
+ // A rolling per-frame sequence in EVERY mode: SVGF's temporal
797
+ // accumulation needs unbiased fresh samples each frame (a frozen
798
+ // sequence converges the EMA to a wrong-but-stable value that
799
+ // reads as static dirty-lens grain glued to the screen). The
800
+ // variance estimate below is what keeps rolling noise from
801
+ // shimmering: it tells the à-trous exactly where to blur hard.
802
+ rng_seed(gid.xy, u.size.z);
803
+
804
+ // PT-3 half-res: realtime mode traces a half grid; map this trace
805
+ // cell to its full-res G-buffer pixel (the 2x2 phase rotates per
806
+ // frame so the EMA integrates all four over time). Full-res modes
807
+ // have ext == size and phase 0, making this the identity.
808
+ var px_full = px;
809
+ if (u.ext.x > u.size.x) {
810
+ // Generalized ratio (the trace grid is budget-capped, so the
811
+ // factor is not necessarily 2): integer scale keeps each trace
812
+ // texel pinned to one owner pixel.
813
+ px_full = min(
814
+ vec2<i32>(
815
+ px.x * i32(u.ext.x) / i32(u.size.x),
816
+ px.y * i32(u.ext.y) / i32(u.size.y),
817
+ ),
818
+ vec2<i32>(i32(u.ext.x) - 1, i32(u.ext.y) - 1),
819
+ );
820
+ }
821
+
822
+ let debug = u.cfg.w;
823
+ if (debug == 5.0) {
824
+ textureStore(out_hdr, px_full, vec4<f32>(1.0, 0.0, 1.0, 1.0));
825
+ return;
826
+ }
827
+
828
+ let depth = depth_at(px_full);
829
+ if (debug == 1.0) {
830
+ textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(depth), 1.0));
831
+ return;
832
+ }
833
+
834
+ if (is_sky(depth)) {
835
+ // Leave the raster sky/clouds untouched. Realtime mode marks
836
+ // the texel as sky in the MOMENTS buffer (depth channel = far
837
+ // plane) so the a-trous passes and the upsampler skip it.
838
+ if (u.cfg.x >= 2.0 && u.cfg.w == 0.0) {
839
+ let sky_idx = gid.y * u.size.x + gid.x;
840
+ accum_out[sky_idx] = vec4<f32>(0.0);
841
+ moments_out[sky_idx] = vec4<f32>(0.0, 0.0, 0.0, 1.0);
842
+ }
843
+ return;
844
+ }
845
+
846
+ let p0 = world_at(px_full, depth);
847
+ let n0 = normal_from_depth(px_full, p0);
848
+ let albedo0 = textureLoad(albedo_tex, px_full, 0).rgb;
849
+
850
+ if (debug == 2.0) {
851
+ textureStore(out_hdr, px_full, vec4<f32>(n0 * 0.5 + 0.5, 1.0));
852
+ return;
853
+ }
854
+ if (debug == 3.0) {
855
+ textureStore(out_hdr, px_full, vec4<f32>(albedo0, 1.0));
856
+ return;
857
+ }
858
+ if (debug == 4.0) {
859
+ let sd = sun_cone_sample(rand_2f());
860
+ let vis = select(0.0, 1.0, !occluded(p0 + n0 * 0.02, sd, 1000.0));
861
+ textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(vis), 1.0));
862
+ return;
863
+ }
864
+ if (debug == 8.0) {
865
+ // Binary probe: white = traced hit has a geometry window,
866
+ // black = geo.z reads 0, red = TLAS miss. HDR-large values so
867
+ // exposure/tonemap can't blur the verdict.
868
+ let dir0 = normalize(p0 - u.cam_pos.xyz);
869
+ var rq8: ray_query;
870
+ rayQueryInitialize(&rq8, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
871
+ loop {
872
+ if (!rayQueryProceed(&rq8)) { break; }
873
+ }
874
+ let h8 = rayQueryGetCommittedIntersection(&rq8);
875
+ var c8 = vec3<f32>(100.0, 0.0, 0.0);
876
+ if (h8.kind != RAY_QUERY_INTERSECTION_NONE) {
877
+ let gi = instance_data[h8.instance_custom_data].geo;
878
+ c8 = select(vec3<f32>(0.0), vec3<f32>(100.0), gi.z > 0u);
879
+ }
880
+ textureStore(out_hdr, px_full, vec4<f32>(c8, 1.0));
881
+ return;
882
+ }
883
+ if (debug == 9.0) {
884
+ // Quantized normal probe: dominant axis of the interpolated
885
+ // world normal as six saturated HDR colours. +X red, -X dark
886
+ // red-ish magenta, +Y green, -Y cyan, +Z blue, -Z yellow.
887
+ // Gray = TLAS miss / no window / zero-length normal.
888
+ let dir0 = normalize(p0 - u.cam_pos.xyz);
889
+ var rq9: ray_query;
890
+ rayQueryInitialize(&rq9, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
891
+ loop {
892
+ if (!rayQueryProceed(&rq9)) { break; }
893
+ }
894
+ let h9 = rayQueryGetCommittedIntersection(&rq9);
895
+ var c9 = vec3<f32>(5.0, 5.0, 5.0);
896
+ if (h9.kind != RAY_QUERY_INTERSECTION_NONE) {
897
+ let inst9 = instance_data[h9.instance_custom_data];
898
+ if (inst9.geo.z > 0u) {
899
+ let a9 = fetch_hit_attrs(inst9.geo, h9.primitive_index, h9.barycentrics);
900
+ let raw = a9.normal_os;
901
+ if (length(raw) > 1e-6) {
902
+ let n9 = normal_to_world(raw, h9.world_to_object);
903
+ let an = abs(n9);
904
+ if (an.y >= an.x && an.y >= an.z) {
905
+ c9 = select(vec3<f32>(0.0, 50.0, 50.0), vec3<f32>(0.0, 50.0, 0.0), n9.y >= 0.0);
906
+ } else if (an.x >= an.z) {
907
+ c9 = select(vec3<f32>(50.0, 0.0, 25.0), vec3<f32>(50.0, 0.0, 0.0), n9.x >= 0.0);
908
+ } else {
909
+ c9 = select(vec3<f32>(50.0, 50.0, 0.0), vec3<f32>(0.0, 0.0, 50.0), n9.z >= 0.0);
910
+ }
911
+ }
912
+ }
913
+ }
914
+ textureStore(out_hdr, px_full, vec4<f32>(c9, 1.0));
915
+ return;
916
+ }
917
+ if (debug == 10.0) {
918
+ // primitive_index sanity probe: banded pseudo-colour of the hit
919
+ // triangle index. Expected: per-triangle colour noise across
920
+ // meshes. A single flat colour everywhere = the field is
921
+ // constant; saturated white = garbage-huge.
922
+ let dir0 = normalize(p0 - u.cam_pos.xyz);
923
+ var rq10: ray_query;
924
+ rayQueryInitialize(&rq10, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
925
+ loop {
926
+ if (!rayQueryProceed(&rq10)) { break; }
927
+ }
928
+ let h10 = rayQueryGetCommittedIntersection(&rq10);
929
+ var c10 = vec3<f32>(0.0);
930
+ if (h10.kind != RAY_QUERY_INTERSECTION_NONE) {
931
+ let prim = f32(h10.primitive_index);
932
+ c10 = vec3<f32>(fract(prim / 64.0), fract(prim / 1024.0), fract(prim / 16384.0)) * 30.0;
933
+ }
934
+ textureStore(out_hdr, px_full, vec4<f32>(c10, 1.0));
935
+ return;
936
+ }
937
+ if (debug == 11.0 || debug == 12.0) {
938
+ // 11: instance_custom_data palette (expect distinct colours per
939
+ // proxy: terrain vs trees vs building). Constant = broken.
940
+ // 12: raw barycentrics (expect smooth per-triangle gradients).
941
+ let dir0 = normalize(p0 - u.cam_pos.xyz);
942
+ var rq11: ray_query;
943
+ rayQueryInitialize(&rq11, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
944
+ loop {
945
+ if (!rayQueryProceed(&rq11)) { break; }
946
+ }
947
+ let h11 = rayQueryGetCommittedIntersection(&rq11);
948
+ var c11 = vec3<f32>(0.0);
949
+ if (h11.kind != RAY_QUERY_INTERSECTION_NONE) {
950
+ if (debug == 11.0) {
951
+ let id = h11.instance_custom_data;
952
+ c11 = vec3<f32>(
953
+ f32((id * 37u) % 7u) / 7.0,
954
+ f32((id * 61u) % 11u) / 11.0,
955
+ f32((id * 13u) % 5u) / 5.0,
956
+ ) * 30.0;
957
+ } else {
958
+ let b = h11.barycentrics;
959
+ c11 = vec3<f32>(b.x, b.y, max(0.0, 1.0 - b.x - b.y)) * 30.0;
960
+ }
961
+ }
962
+ textureStore(out_hdr, px_full, vec4<f32>(c11, 1.0));
963
+ return;
964
+ }
965
+ if (debug == 13.0) {
966
+ // TLAS sanity: green = traced primary hit distance agrees with
967
+ // the G-buffer depth (within 2% + 0.1m), red = disagreement
968
+ // (wrong geometry committed), blue = TLAS miss on a G-buffer
969
+ // pixel. If this is red/blue everywhere the TLAS itself (not
970
+ // the intersection attributes) is broken on this backend.
971
+ let to_p = p0 - u.cam_pos.xyz;
972
+ let gdist = length(to_p);
973
+ let dir0 = to_p / max(gdist, 1e-4);
974
+ var rq13: ray_query;
975
+ rayQueryInitialize(&rq13, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
976
+ loop {
977
+ if (!rayQueryProceed(&rq13)) { break; }
978
+ }
979
+ let h13 = rayQueryGetCommittedIntersection(&rq13);
980
+ var c13 = vec3<f32>(0.0, 0.0, 50.0);
981
+ if (h13.kind != RAY_QUERY_INTERSECTION_NONE) {
982
+ let err = abs(h13.t - gdist);
983
+ if (err < gdist * 0.02 + 0.1) {
984
+ c13 = vec3<f32>(0.0, 50.0, 0.0);
985
+ } else {
986
+ c13 = vec3<f32>(50.0, 0.0, 0.0);
987
+ }
988
+ }
989
+ textureStore(out_hdr, px_full, vec4<f32>(c13, 1.0));
990
+ return;
991
+ }
992
+ if (debug == 14.0) {
993
+ // Shape probe: contour bands of the traced primary hit distance
994
+ // — shows what world the TLAS actually contains. Blue = miss.
995
+ let dir0 = normalize(p0 - u.cam_pos.xyz);
996
+ var rq14: ray_query;
997
+ rayQueryInitialize(&rq14, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
998
+ loop {
999
+ if (!rayQueryProceed(&rq14)) { break; }
1000
+ }
1001
+ let h14 = rayQueryGetCommittedIntersection(&rq14);
1002
+ var c14 = vec3<f32>(0.0, 0.0, 30.0);
1003
+ if (h14.kind != RAY_QUERY_INTERSECTION_NONE) {
1004
+ c14 = vec3<f32>(
1005
+ fract(h14.t * 0.125),
1006
+ fract(h14.t * 0.03125),
1007
+ fract(h14.t * 0.0078125),
1008
+ ) * 20.0;
1009
+ }
1010
+ textureStore(out_hdr, px_full, vec4<f32>(c14, 1.0));
1011
+ return;
1012
+ }
1013
+ if (debug == 15.0) {
1014
+ // Aliasing probe: two queries, two very different rays.
1015
+ // A = primary (per-pixel), B = straight down (t ~= camera
1016
+ // height, near-constant). R channel = banded tA, G = banded tB.
1017
+ // If R == G everywhere the two queries alias to one object.
1018
+ let dirA = normalize(p0 - u.cam_pos.xyz);
1019
+ var rqA: ray_query;
1020
+ rayQueryInitialize(&rqA, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dirA));
1021
+ loop {
1022
+ if (!rayQueryProceed(&rqA)) { break; }
1023
+ }
1024
+ var rqB: ray_query;
1025
+ rayQueryInitialize(&rqB, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, vec3<f32>(0.0, -1.0, 0.0)));
1026
+ loop {
1027
+ if (!rayQueryProceed(&rqB)) { break; }
1028
+ }
1029
+ let hA = rayQueryGetCommittedIntersection(&rqA);
1030
+ let hB = rayQueryGetCommittedIntersection(&rqB);
1031
+ var tA = -1.0;
1032
+ var tB = -1.0;
1033
+ if (hA.kind != RAY_QUERY_INTERSECTION_NONE) { tA = hA.t; }
1034
+ if (hB.kind != RAY_QUERY_INTERSECTION_NONE) { tB = hB.t; }
1035
+ let c15 = vec3<f32>(fract(tA * 0.125) * 20.0, fract(tB * 0.125) * 20.0, 0.0);
1036
+ textureStore(out_hdr, px_full, vec4<f32>(c15, 1.0));
1037
+ return;
1038
+ }
1039
+ if (debug == 16.0) {
1040
+ // Raw numeric dump: traced primary intersection into the accum
1041
+ // buffer as (t, instance_custom_data, primitive_index, kind).
1042
+ // The CPU side reads a window of this buffer back and writes a
1043
+ // text file — no tonemap guesswork.
1044
+ let dir0 = normalize(p0 - u.cam_pos.xyz);
1045
+ var rq16: ray_query;
1046
+ rayQueryInitialize(&rq16, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
1047
+ loop {
1048
+ if (!rayQueryProceed(&rq16)) { break; }
1049
+ }
1050
+ let h16 = rayQueryGetCommittedIntersection(&rq16);
1051
+ let idx16 = gid.y * u.size.x + gid.x;
1052
+ accum_out[idx16] = vec4<f32>(
1053
+ h16.t,
1054
+ f32(h16.instance_custom_data),
1055
+ f32(h16.primitive_index),
1056
+ f32(h16.kind),
1057
+ );
1058
+ textureStore(out_hdr, px_full, vec4<f32>(0.2, 0.0, 0.4, 1.0));
1059
+ return;
1060
+ }
1061
+ if (debug == 17.0) {
1062
+ // Raw ray-generation dump: reconstructed world position + raw
1063
+ // depth, straight into accum for CPU readback.
1064
+ let idx17 = gid.y * u.size.x + gid.x;
1065
+ accum_out[idx17] = vec4<f32>(p0, depth);
1066
+ textureStore(out_hdr, px_full, vec4<f32>(0.4, 0.2, 0.0, 1.0));
1067
+ return;
1068
+ }
1069
+ if (debug == 18.0) {
1070
+ // Hybrid-sun validation: cascade shadow visibility at the
1071
+ // primary surface. Must match the raster shadow shapes exactly
1072
+ // (crisp tree/building shadows). Gray everywhere = the VPs are
1073
+ // wrong (transposition or stale).
1074
+ let vis18 = sun_vis_cascade(p0 + n0 * 0.02);
1075
+ textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(vis18 * 50.0), 1.0));
1076
+ return;
1077
+ }
1078
+ if (debug == 19.0) {
1079
+ // Numeric dump: cascade-0 shadow projection (ndc.xyz) + the
1080
+ // stored atlas depth at the landing texel (-1 = out of range,
1081
+ // -9 = degenerate w). Read back via pt_trace_dump.txt.
1082
+ let pw19 = p0 + n0 * 0.02;
1083
+ let clip19 = u.shadow_vps[0] * vec4<f32>(pw19, 1.0);
1084
+ var out19 = vec4<f32>(-9.0);
1085
+ if (abs(clip19.w) > 1e-6) {
1086
+ let ndc19 = clip19.xyz / clip19.w;
1087
+ let uv19 = vec2<f32>(ndc19.x * 0.5 + 0.5, 0.5 - ndc19.y * 0.5);
1088
+ var stored19 = -1.0;
1089
+ if (uv19.x >= 0.0 && uv19.x <= 1.0 && uv19.y >= 0.0 && uv19.y <= 1.0) {
1090
+ let dims19 = vec2<f32>(textureDimensions(shadow_atlas_0));
1091
+ stored19 = textureLoad(shadow_atlas_0, vec2<i32>(uv19 * dims19), 0);
1092
+ }
1093
+ out19 = vec4<f32>(ndc19.xyz, stored19);
1094
+ }
1095
+ accum_out[gid.y * u.size.x + gid.x] = out19;
1096
+ textureStore(out_hdr, px_full, vec4<f32>(0.1, 0.0, 0.2, 1.0));
1097
+ return;
1098
+ }
1099
+ if (debug == 6.0 || debug == 7.0) {
1100
+ // PT-2 validation: trace the primary ray through the TLAS
1101
+ // (ignoring the G-buffer) and show interpolated attributes.
1102
+ // Should match debug 2/3 up to smooth-vs-screen normals and
1103
+ // card-vs-texture resolution.
1104
+ let dir0 = normalize(p0 - u.cam_pos.xyz);
1105
+ var rq0: ray_query;
1106
+ rayQueryInitialize(&rq0, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
1107
+ loop {
1108
+ if (!rayQueryProceed(&rq0)) { break; }
1109
+ }
1110
+ let h = rayQueryGetCommittedIntersection(&rq0);
1111
+ var col = vec3<f32>(1.0, 0.0, 1.0); // magenta: TLAS miss
1112
+ if (h.kind != RAY_QUERY_INTERSECTION_NONE) {
1113
+ let hinst = instance_data[h.instance_custom_data];
1114
+ if (hinst.geo.z > 0u) {
1115
+ let attrs = fetch_hit_attrs(hinst.geo, h.primitive_index, h.barycentrics);
1116
+ if (debug == 6.0) {
1117
+ col = normal_to_world(attrs.normal_os, h.world_to_object) * 0.5 + vec3<f32>(0.5);
1118
+ } else if (PT_HAS_TEXTURES) {
1119
+ col = hinst.albedo * pt_tex_sample(hinst.geo.w, attrs.uv);
1120
+ } else {
1121
+ col = vec3<f32>(1.0, 1.0, 0.0); // yellow: no tex arrays
1122
+ }
1123
+ } else {
1124
+ col = vec3<f32>(1.0, 0.5, 0.0); // orange: no geo window
1125
+ }
1126
+ }
1127
+ textureStore(out_hdr, px_full, vec4<f32>(col, 1.0));
1128
+ return;
1129
+ }
1130
+
1131
+ // ---- one path sample --------------------------------------------------
1132
+
1133
+ // Primary surface material from the G-buffer (R = metallic,
1134
+ // G = roughness). NEE stays diffuse-only, so scale it by
1135
+ // (1 - metallic) — metals have no diffuse lobe. Specular NEE is a
1136
+ // known gap (see the PT-2 ticket); specular reflection of sky and
1137
+ // scene comes from the GGX bounce below.
1138
+ let mr0 = textureLoad(material_tex, px_full, 0).rg;
1139
+ var metal_cur = mr0.r;
1140
+ var rough_cur = mr0.g;
1141
+ // Realtime mode samples the primary sun cone with structured IGN
1142
+ // noise, rolling per frame: spatially well-distributed (a 5x5
1143
+ // filter averages it nearly flat, unlike white PCG noise) and
1144
+ // temporally unbiased so the SVGF accumulation converges to the
1145
+ // true mean. Under the hybrid cascade sun this path only matters
1146
+ // when shadow maps are disabled. Progressive keeps white noise.
1147
+ var sun_r2 = rand_2f();
1148
+ if (u.cfg.x >= 2.0) {
1149
+ sun_r2 = vec2<f32>(
1150
+ ign_at(px_full, u.size.z),
1151
+ ign_at(px_full + vec2<i32>(17, 59), u.size.z),
1152
+ );
1153
+ }
1154
+ var view_cur = normalize(u.cam_pos.xyz - p0);
1155
+ // Reproject this surface into the previous trace grid ONCE — the
1156
+ // ReSTIR temporal reuse and the SVGF colour accumulation below both
1157
+ // consume the result (rp_* privates).
1158
+ compute_reproj(p0, px_full, depth);
1159
+ // PT-4 experimental: ext.w routes the primary point-light NEE
1160
+ // through the ReSTIR reservoirs; sun NEE is untouched either way.
1161
+ let use_restir = u.ext.w == 1u && u.cfg.x >= 2.0;
1162
+ // Sun + point lights, diffuse AND specular (nee_spec inside) — the
1163
+ // GGX highlight rides the same visibility as the diffuse term.
1164
+ var radiance = direct_light(
1165
+ p0 + n0 * 0.02, n0, albedo0 * (1.0 - metal_cur), sun_r2,
1166
+ view_cur, albedo0, rough_cur, metal_cur, !use_restir,
1167
+ );
1168
+ if (use_restir) {
1169
+ radiance += restir_point_light(
1170
+ gid.y * u.size.x + gid.x, p0 + n0 * 0.02, n0, view_cur,
1171
+ albedo0 * (1.0 - metal_cur), albedo0, rough_cur, metal_cur,
1172
+ );
1173
+ }
1174
+ var throughput = vec3<f32>(1.0);
1175
+ var origin = p0 + n0 * 0.02;
1176
+ var n_cur = n0;
1177
+ var alb_cur = albedo0;
1178
+
1179
+ let max_bounces = u32(u.cfg.y);
1180
+ for (var b = 0u; b < max_bounces; b = b + 1u) {
1181
+ let s = sample_brdf(n_cur, view_cur, alb_cur, rough_cur, metal_cur);
1182
+ if (!s.valid) {
1183
+ break;
1184
+ }
1185
+ throughput *= s.weight;
1186
+ let dir = s.dir;
1187
+
1188
+ var rq: ray_query;
1189
+ rayQueryInitialize(&rq, accel, RayDesc(0u, 0xFFu, 0.001, 500.0, origin, dir));
1190
+ loop {
1191
+ if (!rayQueryProceed(&rq)) { break; }
1192
+ }
1193
+ let hit = rayQueryGetCommittedIntersection(&rq);
1194
+
1195
+ if (hit.kind == RAY_QUERY_INTERSECTION_NONE) {
1196
+ radiance += throughput * sky_radiance(dir);
1197
+ break;
1198
+ }
1199
+
1200
+ let inst = instance_data[hit.instance_custom_data];
1201
+ let hit_ws = origin + dir * hit.t;
1202
+ let hit_os = (hit.world_to_object * vec4<f32>(hit_ws, 1.0)).xyz;
1203
+
1204
+ // PT-2: interpolated vertex normal + textured albedo when the
1205
+ // instance carries a geometry window; PT-1 flat-normal/card
1206
+ // fallback otherwise.
1207
+ var n_hit: vec3<f32>;
1208
+ var alb_hit: vec3<f32>;
1209
+ if (inst.geo.z > 0u) {
1210
+ let attrs = fetch_hit_attrs(inst.geo, hit.primitive_index, hit.barycentrics);
1211
+ n_hit = normal_to_world(attrs.normal_os, hit.world_to_object);
1212
+ if (PT_HAS_TEXTURES) {
1213
+ alb_hit = inst.albedo * pt_tex_sample(inst.geo.w, attrs.uv);
1214
+ } else {
1215
+ alb_hit = albedo_at_hit(inst, hit_os, dir);
1216
+ }
1217
+ } else {
1218
+ var nf = inst.normal_ws;
1219
+ let n_len = length(nf);
1220
+ if (n_len < 1e-4) { nf = -dir; } else { nf = nf / n_len; }
1221
+ n_hit = nf;
1222
+ alb_hit = albedo_at_hit(inst, hit_os, dir);
1223
+ }
1224
+ // A backface (or a flat normal pointing away) still bounces
1225
+ // outward, matching the OPAQUE two-sided raster convention.
1226
+ if (dot(n_hit, dir) > 0.0) { n_hit = -n_hit; }
1227
+
1228
+ // Emissive surfaces radiate; matches the Lumen fallback semantics
1229
+ // (albedo * emissive_luma).
1230
+ radiance += throughput * inst.albedo * inst.emissive_luma;
1231
+
1232
+ let hit_p = hit_ws + n_hit * 0.02;
1233
+ // view at a bounce vertex = back along the incoming ray.
1234
+ // Bounce vertices always use plain NEE (reservoirs are per
1235
+ // PRIMARY texel; reusing them off-surface would be biased).
1236
+ radiance += throughput * direct_light(
1237
+ hit_p, n_hit, alb_hit * (1.0 - inst.mat_params.y), rand_2f(),
1238
+ -dir, alb_hit, inst.mat_params.x, inst.mat_params.y, true,
1239
+ );
1240
+
1241
+ origin = hit_p;
1242
+ n_cur = n_hit;
1243
+ alb_cur = alb_hit;
1244
+ rough_cur = inst.mat_params.x;
1245
+ metal_cur = inst.mat_params.y;
1246
+ view_cur = -dir;
1247
+
1248
+ // Russian roulette from the third bounce.
1249
+ if (b >= 2u) {
1250
+ let q = clamp(max(throughput.r, max(throughput.g, throughput.b)), 0.05, 0.95);
1251
+ if (rand_f() > q) { break; }
1252
+ throughput /= q;
1253
+ }
1254
+ }
1255
+
1256
+ // NaN/Inf guard so one bad sample cannot poison the accumulator.
1257
+ if (radiance.r != radiance.r || radiance.g != radiance.g || radiance.b != radiance.b) {
1258
+ radiance = vec3<f32>(0.0);
1259
+ }
1260
+
1261
+ // ---- accumulate ---------------------------------------------------------
1262
+
1263
+ let idx = gid.y * u.size.x + gid.x;
1264
+ let mode = u.cfg.x;
1265
+ var prev = accum[idx];
1266
+ if (u.size.w == 0u) { prev = vec4<f32>(0.0); }
1267
+
1268
+ var out: vec3<f32>;
1269
+ if (mode >= 2.0) {
1270
+ // SVGF temporal accumulation (Schied et al. 2017). History and
1271
+ // output store IRRADIANCE (radiance demodulated by the primary
1272
+ // albedo) so the wavelet passes filter lighting only; the
1273
+ // final pass re-multiplies by the full-res G-buffer albedo.
1274
+ var irr = radiance / max(albedo0, vec3<f32>(0.05));
1275
+ // Firefly clamp — the one practical deviation from the paper
1276
+ // (reference implementations keep one too): it must bind in
1277
+ // IRRADIANCE space, because dividing by a dark albedo
1278
+ // amplifies radiance outliers up to 20x. Sunlit irradiance
1279
+ // sits around 1-3; 4 leaves real highlights alone.
1280
+ let irr_luma = dot(irr, vec3<f32>(0.2126, 0.7152, 0.0722));
1281
+ if (irr_luma > 4.0) {
1282
+ irr *= 4.0 / irr_luma;
1283
+ }
1284
+ let l_new = min(irr_luma, 4.0);
1285
+
1286
+ // Reprojection: 2x2 BILINEAR taps around the reprojected
1287
+ // position, each tap validated for geometric consistency
1288
+ // (relative linearized depth against the moments buffer).
1289
+ // Point sampling here quantizes to whole trace texels and
1290
+ // forced the old loose-tolerance workaround; weighted taps
1291
+ // give sub-texel reprojection and a honest per-tap test.
1292
+ var hist_rgb = vec3<f32>(0.0);
1293
+ var hist_m1 = 0.0;
1294
+ var hist_m2 = 0.0;
1295
+ var hist_n = 0.0;
1296
+ var wsum = 0.0;
1297
+ // Footprint depth window (current frame): one trace texel
1298
+ // covers ~3x3 full-res pixels; used below to tell a jitter
1299
+ // surface-flip apart from a true disocclusion.
1300
+ var fp_lo = 1e30;
1301
+ var fp_hi = 0.0;
1302
+ {
1303
+ let rx = max(i32(u.ext.x) / i32(u.size.x), 1);
1304
+ let ry = max(i32(u.ext.y) / i32(u.size.y), 1);
1305
+ for (var sy = 0; sy <= 1; sy = sy + 1) {
1306
+ for (var sx = 0; sx <= 1; sx = sx + 1) {
1307
+ let sp = min(
1308
+ px_full + vec2<i32>(sx * (rx - 1), sy * (ry - 1)),
1309
+ vec2<i32>(i32(u.ext.x) - 1, i32(u.ext.y) - 1),
1310
+ );
1311
+ let dz = depth_at(sp);
1312
+ if (dz >= 0.9999999) { continue; }
1313
+ let zl = lin_depth(dz);
1314
+ fp_lo = min(fp_lo, zl);
1315
+ fp_hi = max(fp_hi, zl);
1316
+ }
1317
+ }
1318
+ }
1319
+ // Reprojection basis was computed once at the top of the frame
1320
+ // (rp_* privates, shared with the ReSTIR temporal reuse).
1321
+ if (rp_valid && debug == 23.0) {
1322
+ // Reprojection dump: where this texel thinks it was
1323
+ // last frame (trace-grid units) + the depth pair the
1324
+ // acceptance test compares. Static camera => pos
1325
+ // must equal the texel's own coordinates.
1326
+ accum_out[idx] = vec4<f32>(
1327
+ f32(rp_base.x) + rp_fr.x, f32(rp_base.y) + rp_fr.y,
1328
+ rp_zl_here, lin_depth(moments[rp_nearest].w));
1329
+ moments_out[idx] = vec4<f32>(0.0, 0.0, 0.0, depth);
1330
+ return;
1331
+ }
1332
+ if (rp_valid) {
1333
+ // Tap test is TIGHT (surface identity). Cross-surface
1334
+ // blending is the expensive error: it leaks bright
1335
+ // blade-top lighting onto the ground below (gray-blue
1336
+ // mottle). Surface flips are handled after the loop,
1337
+ // not by widening this tolerance.
1338
+ let tol = 0.1 * rp_zl_here + 0.02;
1339
+ for (var ty = 0; ty <= 1; ty = ty + 1) {
1340
+ for (var tx = 0; tx <= 1; tx = tx + 1) {
1341
+ let q = rp_base + vec2<i32>(tx, ty);
1342
+ if (q.x < 0 || q.y < 0 || q.x >= i32(u.size.x) || q.y >= i32(u.size.y)) {
1343
+ continue;
1344
+ }
1345
+ let qidx = u32(q.y) * u.size.x + u32(q.x);
1346
+ let m = moments[qidx];
1347
+ // Sky texels and depth-inconsistent taps carry
1348
+ // another surface's lighting — skip them.
1349
+ if (m.w >= 0.9999999) {
1350
+ continue;
1351
+ }
1352
+ let zl_hist = lin_depth(m.w);
1353
+ if (abs(zl_hist - rp_zl_here) > tol) {
1354
+ continue;
1355
+ }
1356
+ let wx = mix(1.0 - rp_fr.x, rp_fr.x, f32(tx));
1357
+ let wy = mix(1.0 - rp_fr.y, rp_fr.y, f32(ty));
1358
+ let wt = wx * wy + 1e-4;
1359
+ hist_rgb += accum[qidx].rgb * wt;
1360
+ hist_m1 += m.x * wt;
1361
+ hist_m2 += m.y * wt;
1362
+ hist_n += m.z * wt;
1363
+ wsum += wt;
1364
+ }
1365
+ }
1366
+ }
1367
+
1368
+ // No tap matched. One trace texel holds ONE surface's history;
1369
+ // TAA jitter re-picks which surface the owner pixel sees each
1370
+ // frame on sub-texel geometry (grass). If the STORED surface
1371
+ // still exists in this texel's current footprint, this frame's
1372
+ // sample simply belongs to the other surface: keep the history
1373
+ // verbatim and drop the sample (blending would leak lighting
1374
+ // across surfaces; resetting would pin the texel at 1 spp and
1375
+ // fill the screen with speckle). The upsampler routes trace
1376
+ // texels to full-res pixels by depth, so the other surface
1377
+ // draws its lighting from neighbouring texels. Only a stored
1378
+ // surface that has LEFT the footprint is a true disocclusion.
1379
+ if (wsum <= 1e-3 && rp_valid) {
1380
+ let mnp = moments[rp_nearest];
1381
+ if (mnp.w < 0.9999999 && mnp.z > 0.0 && fp_hi > 0.0 && fp_lo < 1e29) {
1382
+ let zl_st = lin_depth(mnp.w);
1383
+ let wtol = 0.1 * zl_st + 0.02;
1384
+ if (zl_st > fp_lo - wtol && zl_st < fp_hi + wtol) {
1385
+ accum_out[idx] = accum[rp_nearest];
1386
+ moments_out[idx] = mnp;
1387
+ if (debug == 22.0) {
1388
+ // Numeric dump: n / variance / 99 = flip path / depth.
1389
+ accum_out[idx] = vec4<f32>(mnp.z, max(mnp.y - mnp.x * mnp.x, 0.0), 99.0, mnp.w);
1390
+ }
1391
+ if (debug == 20.0) {
1392
+ textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(mnp.z / 32.0), 1.0));
1393
+ }
1394
+ if (debug == 21.0) {
1395
+ let v_st = max(mnp.y - mnp.x * mnp.x, 0.0);
1396
+ textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(v_st * 10.0), 1.0));
1397
+ }
1398
+ return;
1399
+ }
1400
+ }
1401
+ }
1402
+
1403
+ var n_hist = 0.0;
1404
+ var seeded = false;
1405
+ if (wsum > 1e-3) {
1406
+ hist_rgb /= wsum;
1407
+ hist_m1 /= wsum;
1408
+ hist_m2 /= wsum;
1409
+ n_hist = hist_n / wsum;
1410
+ } else {
1411
+ // PT-9 — disocclusion seeding. A true disocclusion used to start
1412
+ // from the raw 1-spp sample with variance EXACTLY zero (m2 - m1²
1413
+ // of a single value), so freshly streamed-in pixels — the whole
1414
+ // viewport periphery whenever the camera translates — showed raw
1415
+ // path noise until history rebuilt. Indoors (bounce-dominated
1416
+ // light) that read as a salt-and-pepper ring around a clean
1417
+ // centre. But a newborn's NEIGHBOURS usually hold converged
1418
+ // history for the very same surface (the wall streaming in at
1419
+ // the screen edge continues inward), and accum[]/moments[] are
1420
+ // the previous frame's buffers — race-free to read at any texel.
1421
+ // Borrow from depth-consistent, converged (n ≥ 4) neighbours,
1422
+ // weighted by their history length; inherit HALF their history
1423
+ // (capped at 8) so the canonical 1/N blend still folds real
1424
+ // fresh samples in quickly. A pixel with no depth-compatible
1425
+ // neighbour (a genuinely new surface, e.g. rounding a doorway)
1426
+ // keeps the honest raw start — there is nothing to borrow.
1427
+ let zl_here = lin_depth(depth);
1428
+ let btol = 0.1 * zl_here + 0.02;
1429
+ var seed_rgb = vec3<f32>(0.0);
1430
+ var seed_m1 = 0.0;
1431
+ var seed_m2 = 0.0;
1432
+ var seed_n = 0.0;
1433
+ var seed_w = 0.0;
1434
+ // 7x7: under TRANSLATION a whole COLUMN of texels streams in per
1435
+ // frame (rotation only trickles a few px), so a 5x5 often found
1436
+ // nothing but fellow newborns and the band stayed raw.
1437
+ for (var by = -3; by <= 3; by = by + 1) {
1438
+ for (var bx = -3; bx <= 3; bx = bx + 1) {
1439
+ if (bx == 0 && by == 0) { continue; }
1440
+ let q = vec2<i32>(i32(gid.x) + bx, i32(gid.y) + by);
1441
+ if (q.x < 0 || q.y < 0 || q.x >= i32(u.size.x) || q.y >= i32(u.size.y)) {
1442
+ continue;
1443
+ }
1444
+ let qidx = u32(q.y) * u.size.x + u32(q.x);
1445
+ let m = moments[qidx];
1446
+ if (m.w >= 0.9999999 || m.z < 4.0) { continue; }
1447
+ if (abs(lin_depth(m.w) - zl_here) > btol) { continue; }
1448
+ let wt = m.z;
1449
+ seed_rgb += accum[qidx].rgb * wt;
1450
+ seed_m1 += m.x * wt;
1451
+ seed_m2 += m.y * wt;
1452
+ seed_n += m.z * wt;
1453
+ seed_w += wt;
1454
+ }
1455
+ }
1456
+ if (seed_w > 0.0) {
1457
+ hist_rgb = seed_rgb / seed_w;
1458
+ hist_m1 = seed_m1 / seed_w;
1459
+ hist_m2 = seed_m2 / seed_w;
1460
+ n_hist = min((seed_n / seed_w) * 0.5, 8.0);
1461
+ seeded = true;
1462
+ }
1463
+ }
1464
+ // Canonical blend: cumulative average while the history is
1465
+ // young (alpha = 1/N), settling to EMA. The floor is 0.1
1466
+ // rather than the paper's 0.2: our trace is half-res with a
1467
+ // 2-bounce sky lottery as the dominant noise source, and the
1468
+ // deeper average halves the residual mottle. Direct sun comes
1469
+ // from the raster cascades (deterministic), so the slower EMA
1470
+ // only delays indirect/ambient changes (~10 frames).
1471
+ let n_new = min(n_hist + 1.0, 32.0);
1472
+ let alpha_c = max(1.0 / n_new, 0.1);
1473
+ let out_irr = mix(hist_rgb, irr, alpha_c);
1474
+ let m1 = mix(hist_m1, l_new, alpha_c);
1475
+ let m2 = mix(hist_m2, l_new * l_new, alpha_c);
1476
+ // Temporal luminance variance — the signal that drives the
1477
+ // wavelet filter's luminance sigma. Young history makes this
1478
+ // unreliable; the first à-trous iteration substitutes a
1479
+ // spatial estimate when n < 4 (accum.w carries n via moments).
1480
+ var variance = max(m2 - m1 * m1, 0.0);
1481
+ // A newborn with NOTHING to borrow is a 1-sample estimate whose true
1482
+ // variance is unknown — not zero, which is what m2 - m1² of a single
1483
+ // value degenerates to. Zero variance tells the wavelet the pixel is
1484
+ // CONVERGED, so the raw outlier survived every iteration: that was
1485
+ // the residual noise band on wide stream-ins under camera
1486
+ // translation. Write a frank variance instead so the à-trous blurs
1487
+ // these pixels hard; one converged frame later the real statistics
1488
+ // take over.
1489
+ if (n_new <= 1.5 && !seeded) {
1490
+ variance = max(l_new * l_new, 0.25);
1491
+ }
1492
+ accum_out[idx] = vec4<f32>(out_irr, variance);
1493
+ moments_out[idx] = vec4<f32>(m1, m2, n_new, depth);
1494
+ if (debug == 22.0) {
1495
+ // Numeric dump: n / variance / accepted tap mass / depth.
1496
+ accum_out[idx] = vec4<f32>(n_new, variance, wsum, depth);
1497
+ }
1498
+ if (debug == 20.0) {
1499
+ // History length heat: white = full 32-frame history.
1500
+ textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(n_new / 32.0), 1.0));
1501
+ }
1502
+ if (debug == 21.0) {
1503
+ // Variance view (x10 so typical values are visible).
1504
+ textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(variance * 10.0), 1.0));
1505
+ }
1506
+ return;
1507
+ } else {
1508
+ // Progressive keeps its firefly cap, relaxing with
1509
+ // accumulation depth (a deep average can absorb real energy).
1510
+ let luma = dot(radiance, vec3<f32>(0.2126, 0.7152, 0.0722));
1511
+ let cap = 4.0 + f32(min(u.size.w, 28u));
1512
+ if (luma > cap) { radiance *= cap / luma; }
1513
+ // Progressive: plain running sum; count lives on the CPU.
1514
+ // Ping-pong read/write at the same index (static camera only).
1515
+ let sum = prev.rgb + radiance;
1516
+ let n = f32(u.size.w) + 1.0;
1517
+ accum_out[idx] = vec4<f32>(sum, n);
1518
+ out = sum / n;
1519
+ // Interim gameplay behaviour until PT-3's denoiser: a moving
1520
+ // camera resets accumulation every frame, and raw 1-spp noise
1521
+ // through TSR looks terrible. Keep accumulating but leave the
1522
+ // raster frame on screen until a few samples exist — stand
1523
+ // still for half a second and PT dissolves in. The CPU side
1524
+ // mirrors this threshold (pt_wrote_frame) so SSGI/SSR stay on
1525
+ // for the raster frames.
1526
+ if (u.size.w < 8u) {
1527
+ return;
1528
+ }
1529
+ }
1530
+
1531
+ textureStore(out_hdr, px_full, vec4<f32>(out, 1.0));
1532
+ }
1533
+ "#;
1534
+
1535
+ /// PT-3b — SVGF wavelet filter (Schied et al. 2017) for the realtime
1536
+ /// mode. Four `cs_mid` à-trous iterations (steps 1/2/4/8) run on the
1537
+ /// trace grid over the temporally-accumulated irradiance; `cs_final`
1538
+ /// joint-bilaterally upsamples to full resolution and re-modulates the
1539
+ /// G-buffer albedo. Buffers: src/dst = (irradiance rgb, luminance
1540
+ /// variance w); `geo` = the kernel's moments buffer (mu1, mu2, history
1541
+ /// length, raw depth) — static across iterations, it carries the depth
1542
+ /// for edge-stopping and the sky marker (depth = far plane).
1543
+ ///
1544
+ /// The luminance edge-stop is VARIANCE-DRIVEN: sigma_l scales with the
1545
+ /// per-texel noise estimate, so grainy regions blur hard while
1546
+ /// converged shading detail survives. Variance travels with the signal,
1547
+ /// filtered by the squared weights, shrinking each iteration exactly as
1548
+ /// the residual noise does. This replaces the old fixed sigma schedule,
1549
+ /// the despeckle clamp and the history spike clamp — with a correct
1550
+ /// variance estimate none of those are needed.
1551
+ pub(in crate::renderer) const PT_ATROUS_WGSL: &str = r#"
1552
+ struct AtrousParams {
1553
+ // x = step (texels), y = 1.0 on the FIRST iteration (enables the
1554
+ // short-history spatial variance fallback), z/w = trace dims
1555
+ p: vec4<f32>,
1556
+ // x/y = full G-buffer dims (cs_final upsamples trace -> full when
1557
+ // they differ), z/w unused.
1558
+ p2: vec4<f32>,
1559
+ };
1560
+ @group(0) @binding(0) var<uniform> ap: AtrousParams;
1561
+ @group(0) @binding(1) var<storage, read> src: array<vec4<f32>>;
1562
+ @group(0) @binding(2) var<storage, read_write> dst: array<vec4<f32>>;
1563
+ @group(0) @binding(3) var out_hdr_a: texture_storage_2d<rgba16float, write>;
1564
+ @group(0) @binding(4) var depth_full: texture_depth_2d;
1565
+ // Full-res G-buffer albedo: cs_final re-modulates the filtered
1566
+ // irradiance with it (SVGF demodulation keeps textures crisp).
1567
+ @group(0) @binding(5) var albedo_full: texture_2d<f32>;
1568
+ // Kernel moments buffer: (mu1, mu2, history length, raw depth).
1569
+ @group(0) @binding(6) var<storage, read> geo: array<vec4<f32>>;
1570
+
1571
+ fn lin_depth_a(d: f32) -> f32 {
1572
+ return 0.02 / max(1.0 - d, 1e-6);
1573
+ }
1574
+
1575
+ fn luma_of(c: vec3<f32>) -> f32 {
1576
+ return dot(c, vec3<f32>(0.2126, 0.7152, 0.0722));
1577
+ }
1578
+
1579
+ // B3-spline kernel weight for |offset| 0/1/2.
1580
+ fn kern(d: i32) -> f32 {
1581
+ let a = abs(d);
1582
+ if (a == 0) { return 0.375; }
1583
+ if (a == 1) { return 0.25; }
1584
+ return 0.0625;
1585
+ }
1586
+
1587
+ // SVGF luminance sigma (paper value).
1588
+ const SIGMA_L: f32 = 4.0;
1589
+
1590
+ fn filter_at(px: vec2<i32>, w: i32, h: i32, step: i32, first: bool) -> vec4<f32> {
1591
+ let cidx = u32(px.y) * u32(w) + u32(px.x);
1592
+ let g_c = geo[cidx];
1593
+ let center = src[cidx];
1594
+ if (g_c.w >= 0.9999999) {
1595
+ return center;
1596
+ }
1597
+ let zc = lin_depth_a(g_c.w);
1598
+ let lc = luma_of(center.rgb);
1599
+
1600
+ // Center variance. Temporal variance from a young history (< 4
1601
+ // frames, e.g. right after a disocclusion) is meaningless — the
1602
+ // paper substitutes a spatial luminance-variance estimate there.
1603
+ var var_c = max(center.w, 0.0);
1604
+ if (first && g_c.z < 4.0) {
1605
+ var s1 = 0.0;
1606
+ var s2 = 0.0;
1607
+ var cnt = 0.0;
1608
+ for (var dy = -1; dy <= 1; dy = dy + 1) {
1609
+ for (var dx = -1; dx <= 1; dx = dx + 1) {
1610
+ let q = px + vec2<i32>(dx, dy);
1611
+ if (q.x < 0 || q.y < 0 || q.x >= w || q.y >= h) {
1612
+ continue;
1613
+ }
1614
+ let qi = u32(q.y) * u32(w) + u32(q.x);
1615
+ if (geo[qi].w >= 0.9999999) {
1616
+ continue;
1617
+ }
1618
+ let lq = luma_of(src[qi].rgb);
1619
+ s1 += lq;
1620
+ s2 += lq * lq;
1621
+ cnt += 1.0;
1622
+ }
1623
+ }
1624
+ if (cnt > 1.0) {
1625
+ let mu = s1 / cnt;
1626
+ var_c = max(var_c, max(s2 / cnt - mu * mu, 0.0));
1627
+ }
1628
+ // Variance boost: a young history's estimate is unreliable in
1629
+ // BOTH directions, and underestimating is the expensive error
1630
+ // (a 1-spp outlier with variance 0 survives every iteration as
1631
+ // a false edge). Floor it so fresh texels blend spatially until
1632
+ // their temporal estimate matures.
1633
+ var_c = max(var_c, 0.25);
1634
+ }
1635
+
1636
+ // 3x3 gaussian prefilter of the variance (paper 4.2): keeps a
1637
+ // single hot texel from stopping its own smoothing.
1638
+ var vsum = var_c * 0.25;
1639
+ var vwsum = 0.25;
1640
+ for (var dy = -1; dy <= 1; dy = dy + 1) {
1641
+ for (var dx = -1; dx <= 1; dx = dx + 1) {
1642
+ if (dx == 0 && dy == 0) {
1643
+ continue;
1644
+ }
1645
+ let q = px + vec2<i32>(dx, dy);
1646
+ if (q.x < 0 || q.y < 0 || q.x >= w || q.y >= h) {
1647
+ continue;
1648
+ }
1649
+ let qi = u32(q.y) * u32(w) + u32(q.x);
1650
+ if (geo[qi].w >= 0.9999999) {
1651
+ continue;
1652
+ }
1653
+ let gw = select(0.0625, 0.125, dx == 0 || dy == 0);
1654
+ vsum += max(src[qi].w, 0.0) * gw;
1655
+ vwsum += gw;
1656
+ }
1657
+ }
1658
+ let sigma_l_denom = SIGMA_L * sqrt(max(vsum / vwsum, 0.0)) + 1e-3;
1659
+
1660
+ // Depth gradient (central differences on linear depth) scales the
1661
+ // depth edge-stop so steep-slope surfaces stay connected while
1662
+ // depth discontinuities still stop the filter (paper eq. 3).
1663
+ var dzdx = 0.0;
1664
+ var dzdy = 0.0;
1665
+ if (px.x > 0 && px.x < w - 1) {
1666
+ dzdx = (lin_depth_a(geo[cidx + 1u].w) - lin_depth_a(geo[cidx - 1u].w)) * 0.5;
1667
+ }
1668
+ if (px.y > 0 && px.y < h - 1) {
1669
+ dzdy = (lin_depth_a(geo[cidx + u32(w)].w) - lin_depth_a(geo[cidx - u32(w)].w)) * 0.5;
1670
+ }
1671
+
1672
+ var sum = vec3<f32>(0.0);
1673
+ var sum_v = 0.0;
1674
+ var wsum = 0.0;
1675
+ for (var dy = -2; dy <= 2; dy = dy + 1) {
1676
+ for (var dx = -2; dx <= 2; dx = dx + 1) {
1677
+ let q = px + vec2<i32>(dx, dy) * step;
1678
+ if (q.x < 0 || q.y < 0 || q.x >= w || q.y >= h) {
1679
+ continue;
1680
+ }
1681
+ let qi = u32(q.y) * u32(w) + u32(q.x);
1682
+ let g_q = geo[qi];
1683
+ if (g_q.w >= 0.9999999) {
1684
+ continue;
1685
+ }
1686
+ let s = src[qi];
1687
+ let zq = lin_depth_a(g_q.w);
1688
+ let z_denom = abs(dzdx) * f32(abs(dx) * step)
1689
+ + abs(dzdy) * f32(abs(dy) * step)
1690
+ + 0.01 * zc + 1e-4;
1691
+ let wz = exp(-abs(zq - zc) / z_denom);
1692
+ let wl = exp(-abs(luma_of(s.rgb) - lc) / sigma_l_denom);
1693
+ let wgt = kern(dx) * kern(dy) * wz * wl;
1694
+ sum += s.rgb * wgt;
1695
+ // Variance contracts with the SQUARED weights — the
1696
+ // estimate shrinks exactly as the filtered noise does.
1697
+ sum_v += max(s.w, 0.0) * wgt * wgt;
1698
+ wsum += wgt;
1699
+ }
1700
+ }
1701
+ if (wsum < 1e-6) {
1702
+ return center;
1703
+ }
1704
+ return vec4<f32>(sum / wsum, sum_v / (wsum * wsum));
1705
+ }
1706
+
1707
+ @compute @workgroup_size(8, 8, 1)
1708
+ fn cs_mid(@builtin(global_invocation_id) gid: vec3<u32>) {
1709
+ let w = i32(ap.p.z);
1710
+ let h = i32(ap.p.w);
1711
+ if (i32(gid.x) >= w || i32(gid.y) >= h) {
1712
+ return;
1713
+ }
1714
+ let px = vec2<i32>(i32(gid.x), i32(gid.y));
1715
+ dst[gid.y * u32(w) + gid.x] = filter_at(px, w, h, i32(ap.p.x), ap.p.y > 0.5);
1716
+ }
1717
+
1718
+ // Final pass runs at FULL resolution: depth-guided joint-bilateral
1719
+ // upsample of the filtered irradiance, re-modulated by the full-res
1720
+ // G-buffer albedo. Sky pixels (full-res depth at far plane) are never
1721
+ // written so the raster sky survives. When the trace grid IS the full
1722
+ // grid it degenerates to a plain modulate-and-write.
1723
+ @compute @workgroup_size(8, 8, 1)
1724
+ fn cs_final(@builtin(global_invocation_id) gid: vec3<u32>) {
1725
+ let fw = i32(ap.p2.x);
1726
+ let fh = i32(ap.p2.y);
1727
+ if (i32(gid.x) >= fw || i32(gid.y) >= fh) {
1728
+ return;
1729
+ }
1730
+ let px = vec2<i32>(i32(gid.x), i32(gid.y));
1731
+ let d = textureLoad(depth_full, px, 0);
1732
+ if (d >= 0.9999999) {
1733
+ return;
1734
+ }
1735
+ let hw = i32(ap.p.z);
1736
+ let hh = i32(ap.p.w);
1737
+ let alb = max(textureLoad(albedo_full, px, 0).rgb, vec3<f32>(0.05));
1738
+ if (hw == fw) {
1739
+ let cidx = gid.y * u32(fw) + gid.x;
1740
+ if (geo[cidx].w >= 0.9999999) {
1741
+ return;
1742
+ }
1743
+ textureStore(out_hdr_a, px, vec4<f32>(src[cidx].rgb * alb, 1.0));
1744
+ return;
1745
+ }
1746
+ // 3x3 taps around the trace texel, weighted by SUB-TEXEL tent distance
1747
+ // and relative linear-depth agreement with THIS full-res pixel. The
1748
+ // weights used to centre on the CONTAINING texel (integer mapping, no
1749
+ // fractional phase), which reconstructed the lighting as trace-texel-
1750
+ // sized constant blocks — and because the trace grid's sample phase
1751
+ // rotates every frame, the block boundaries CRAWLED under motion (the
1752
+ // residual "pixelated while moving" indoors). Tent weights at the
1753
+ // continuous position make this a true bilinear-plus-depth joint
1754
+ // bilateral: smooth gradients, stable under the phase rotation. The
1755
+ // epsilon keeps thin foreground geometry (whose taps all mismatch)
1756
+ // softly averaged rather than black.
1757
+ let zc = lin_depth_a(d);
1758
+ // Generalized ratio mapping (trace grid is budget-capped, not
1759
+ // always exactly half of full res), texel centres aligned.
1760
+ let fx = (f32(px.x) + 0.5) * f32(hw) / f32(fw) - 0.5;
1761
+ let fy = (f32(px.y) + 0.5) * f32(hh) / f32(fh) - 0.5;
1762
+ let bx = i32(floor(fx));
1763
+ let by = i32(floor(fy));
1764
+ let frx = fx - f32(bx);
1765
+ let fry = fy - f32(by);
1766
+ var sum = vec3<f32>(0.0);
1767
+ var wsum = 0.0;
1768
+ for (var dy = -1; dy <= 1; dy = dy + 1) {
1769
+ for (var dx = -1; dx <= 1; dx = dx + 1) {
1770
+ let qx = bx + dx;
1771
+ let qy = by + dy;
1772
+ if (qx < 0 || qy < 0 || qx >= hw || qy >= hh) {
1773
+ continue;
1774
+ }
1775
+ let qi = u32(qy) * u32(hw) + u32(qx);
1776
+ if (geo[qi].w >= 0.9999999) {
1777
+ continue;
1778
+ }
1779
+ let s = src[qi];
1780
+ let wx = max(0.0, 1.0 - abs(f32(dx) - frx));
1781
+ let wy = max(0.0, 1.0 - abs(f32(dy) - fry));
1782
+ let wz = exp(-abs(lin_depth_a(geo[qi].w) - zc) / (0.08 * zc + 0.02));
1783
+ let wgt = wx * wy * wz + 1e-5;
1784
+ sum += s.rgb * wgt;
1785
+ wsum += wgt;
1786
+ }
1787
+ }
1788
+ if (wsum < 1e-6) {
1789
+ return;
1790
+ }
1791
+ textureStore(out_hdr_a, px, vec4<f32>((sum / wsum) * alb, 1.0));
1792
+ }
1793
+ "#;
1794
+
1795
+ /// PT-6 — compute pre-skin: poses one skinned mesh into its PT geometry
1796
+ /// megabuffer window (world space; the joint palette bakes placement).
1797
+ /// The CPU wrote the bind-pose vertex data into the window this frame;
1798
+ /// this pass overwrites position + normal, leaving color/uv untouched.
1799
+ /// The BLAS for the dynamic instance then reads the same window
1800
+ /// (first_vertex offset), so intersection and hit shading share bytes.
1801
+ pub(in crate::renderer) const PT_SKIN_WGSL: &str = r#"
1802
+ struct SkinParams {
1803
+ // Places the rare rigid (weightless) verts, same as the raster VS.
1804
+ model: mat4x4<f32>,
1805
+ // x = megabuffer vertex slot base (Vertex3D units), y = vertex
1806
+ // count, z = joint palette base offset, w unused.
1807
+ p: vec4<u32>,
1808
+ };
1809
+ struct SkinJoints {
1810
+ m: array<mat4x4<f32>, 1024>,
1811
+ };
1812
+ @group(0) @binding(0) var<uniform> sp: SkinParams;
1813
+ @group(0) @binding(1) var<storage, read> src_v: array<f32>;
1814
+ @group(0) @binding(2) var<storage, read_write> dst_v: array<f32>;
1815
+ @group(0) @binding(3) var<uniform> joints: SkinJoints;
1816
+
1817
+ @compute @workgroup_size(64, 1, 1)
1818
+ fn cs_skin(@builtin(global_invocation_id) gid: vec3<u32>) {
1819
+ let i = gid.x;
1820
+ if (i >= sp.p.y) { return; }
1821
+ // Vertex3D words: pos +0, normal +3, color +6, uv +10, joints +12,
1822
+ // weights +16, tangent +20 (stride 24 f32).
1823
+ let s = i * 24u;
1824
+ let d = (sp.p.x + i) * 24u;
1825
+ let pos = vec4<f32>(src_v[s], src_v[s + 1u], src_v[s + 2u], 1.0);
1826
+ let nrm = vec4<f32>(src_v[s + 3u], src_v[s + 4u], src_v[s + 5u], 0.0);
1827
+ let w = vec4<f32>(
1828
+ src_v[s + 16u], src_v[s + 17u], src_v[s + 18u], src_v[s + 19u],
1829
+ );
1830
+ var wp: vec3<f32>;
1831
+ var wn: vec3<f32>;
1832
+ if (w.x + w.y + w.z + w.w > 0.01) {
1833
+ // Same palette blend as the raster VS (core.rs vs_main_scene).
1834
+ let j0 = u32(src_v[s + 12u]) + sp.p.z;
1835
+ let j1 = u32(src_v[s + 13u]) + sp.p.z;
1836
+ let j2 = u32(src_v[s + 14u]) + sp.p.z;
1837
+ let j3 = u32(src_v[s + 15u]) + sp.p.z;
1838
+ let m0 = joints.m[j0];
1839
+ let m1 = joints.m[j1];
1840
+ let m2 = joints.m[j2];
1841
+ let m3 = joints.m[j3];
1842
+ wp = ((m0 * pos) * w.x + (m1 * pos) * w.y
1843
+ + (m2 * pos) * w.z + (m3 * pos) * w.w).xyz;
1844
+ wn = ((m0 * nrm) * w.x + (m1 * nrm) * w.y
1845
+ + (m2 * nrm) * w.z + (m3 * nrm) * w.w).xyz;
1846
+ } else {
1847
+ wp = (sp.model * pos).xyz;
1848
+ wn = (sp.model * nrm).xyz;
1849
+ }
1850
+ let ln = length(wn);
1851
+ if (ln > 1e-6) { wn = wn / ln; }
1852
+ dst_v[d] = wp.x;
1853
+ dst_v[d + 1u] = wp.y;
1854
+ dst_v[d + 2u] = wp.z;
1855
+ dst_v[d + 3u] = wn.x;
1856
+ dst_v[d + 4u] = wn.y;
1857
+ dst_v[d + 5u] = wn.z;
1858
+ }
1859
+ "#;