@bornengine/engine 0.4.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (213) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +231 -0
  3. package/native/android/Cargo.lock +1848 -0
  4. package/native/android/Cargo.toml +24 -0
  5. package/native/android/src/lib.rs +702 -0
  6. package/native/ios/Cargo.lock +1690 -0
  7. package/native/ios/Cargo.toml +32 -0
  8. package/native/ios/src/lib.rs +1267 -0
  9. package/native/linux/Cargo.lock +3279 -0
  10. package/native/linux/Cargo.toml +29 -0
  11. package/native/linux/src/lib.rs +1331 -0
  12. package/native/macos/Cargo.lock +3310 -0
  13. package/native/macos/Cargo.toml +46 -0
  14. package/native/macos/src/lib.rs +1302 -0
  15. package/native/shared/Cargo.lock +1899 -0
  16. package/native/shared/Cargo.toml +62 -0
  17. package/native/shared/assets/default_font.ttf +0 -0
  18. package/native/shared/build.rs +270 -0
  19. package/native/shared/shaders/common/clouds.wgsl +122 -0
  20. package/native/shared/shaders/common/fog.wgsl +16 -0
  21. package/native/shared/shaders/common/foliage_wind.wgsl +98 -0
  22. package/native/shared/shaders/common/imposter.wgsl +112 -0
  23. package/native/shared/shaders/common/pbr.wgsl +186 -0
  24. package/native/shared/shaders/common/shadows.wgsl +186 -0
  25. package/native/shared/shaders/common/sky.wgsl +8 -0
  26. package/native/shared/shaders/common/tonemap.wgsl +25 -0
  27. package/native/shared/shaders/impulse_field.wgsl +57 -0
  28. package/native/shared/shaders/material_abi.wgsl +383 -0
  29. package/native/shared/shaders/materials/test_minimal.wgsl +42 -0
  30. package/native/shared/src/anim_mixer.rs +61 -0
  31. package/native/shared/src/attach.rs +263 -0
  32. package/native/shared/src/audio/decode.rs +123 -0
  33. package/native/shared/src/audio/mod.rs +863 -0
  34. package/native/shared/src/audio/render.rs +892 -0
  35. package/native/shared/src/audio/spsc.rs +156 -0
  36. package/native/shared/src/audio/stream.rs +226 -0
  37. package/native/shared/src/custom_shaders.rs +104 -0
  38. package/native/shared/src/decals.rs +245 -0
  39. package/native/shared/src/drs.rs +211 -0
  40. package/native/shared/src/engine.rs +261 -0
  41. package/native/shared/src/ffi.rs +116 -0
  42. package/native/shared/src/ffi_core/assets.rs +388 -0
  43. package/native/shared/src/ffi_core/audio_ffi.rs +184 -0
  44. package/native/shared/src/ffi_core/draw.rs +334 -0
  45. package/native/shared/src/ffi_core/game_loop.rs +577 -0
  46. package/native/shared/src/ffi_core/input.rs +234 -0
  47. package/native/shared/src/ffi_core/mod.rs +127 -0
  48. package/native/shared/src/ffi_core/models.rs +1154 -0
  49. package/native/shared/src/ffi_core/ragdoll_ffi.rs +261 -0
  50. package/native/shared/src/ffi_core/scene.rs +626 -0
  51. package/native/shared/src/ffi_core/vfx.rs +212 -0
  52. package/native/shared/src/ffi_core/visual.rs +691 -0
  53. package/native/shared/src/frame_callbacks.rs +122 -0
  54. package/native/shared/src/geometry.rs +236 -0
  55. package/native/shared/src/handles.rs +182 -0
  56. package/native/shared/src/input.rs +448 -0
  57. package/native/shared/src/jolt_sys.rs +822 -0
  58. package/native/shared/src/lib.rs +55 -0
  59. package/native/shared/src/models.rs +1093 -0
  60. package/native/shared/src/models_gltf.rs +1280 -0
  61. package/native/shared/src/particles.rs +391 -0
  62. package/native/shared/src/physics_jolt.rs +1908 -0
  63. package/native/shared/src/picking.rs +298 -0
  64. package/native/shared/src/postfx.rs +345 -0
  65. package/native/shared/src/profiler.rs +492 -0
  66. package/native/shared/src/ragdoll.rs +474 -0
  67. package/native/shared/src/renderer/atmosphere_lut.rs +573 -0
  68. package/native/shared/src/renderer/brdf_lut.rs +154 -0
  69. package/native/shared/src/renderer/draw2d.rs +143 -0
  70. package/native/shared/src/renderer/formats.rs +822 -0
  71. package/native/shared/src/renderer/froxel.rs +421 -0
  72. package/native/shared/src/renderer/gi_bake.rs +653 -0
  73. package/native/shared/src/renderer/graph.rs +462 -0
  74. package/native/shared/src/renderer/hiz.rs +269 -0
  75. package/native/shared/src/renderer/hot_reload.rs +390 -0
  76. package/native/shared/src/renderer/impulse_field.rs +456 -0
  77. package/native/shared/src/renderer/lighting.rs +154 -0
  78. package/native/shared/src/renderer/material_instancing.rs +171 -0
  79. package/native/shared/src/renderer/material_pipeline.rs +700 -0
  80. package/native/shared/src/renderer/material_system.rs +1996 -0
  81. package/native/shared/src/renderer/material_system_tests.rs +601 -0
  82. package/native/shared/src/renderer/material_system_wasm.rs +41 -0
  83. package/native/shared/src/renderer/mod.rs +12556 -0
  84. package/native/shared/src/renderer/model_draw.rs +641 -0
  85. package/native/shared/src/renderer/occlusion.rs +429 -0
  86. package/native/shared/src/renderer/planar_pass.rs +593 -0
  87. package/native/shared/src/renderer/planar_reflection.rs +499 -0
  88. package/native/shared/src/renderer/post_pass.rs +249 -0
  89. package/native/shared/src/renderer/postfx_chain.rs +728 -0
  90. package/native/shared/src/renderer/pt_pass.rs +577 -0
  91. package/native/shared/src/renderer/scene_pass.rs +607 -0
  92. package/native/shared/src/renderer/shader_include.rs +205 -0
  93. package/native/shared/src/renderer/shader_library.rs +135 -0
  94. package/native/shared/src/renderer/shaders/ao.rs +570 -0
  95. package/native/shared/src/renderer/shaders/core.rs +1243 -0
  96. package/native/shared/src/renderer/shaders/env.rs +907 -0
  97. package/native/shared/src/renderer/shaders/gi.rs +810 -0
  98. package/native/shared/src/renderer/shaders/mod.rs +19 -0
  99. package/native/shared/src/renderer/shaders/post.rs +1558 -0
  100. package/native/shared/src/renderer/shaders/pt.rs +1859 -0
  101. package/native/shared/src/renderer/shaders/ssgi.rs +1586 -0
  102. package/native/shared/src/renderer/shadow_pass.rs +731 -0
  103. package/native/shared/src/renderer/ssgi_pass.rs +392 -0
  104. package/native/shared/src/renderer/ssr_pass.rs +188 -0
  105. package/native/shared/src/renderer/texture_store.rs +473 -0
  106. package/native/shared/src/renderer/transient.rs +591 -0
  107. package/native/shared/src/renderer/types.rs +941 -0
  108. package/native/shared/src/renderer/util.rs +152 -0
  109. package/native/shared/src/scene.rs +1362 -0
  110. package/native/shared/src/sdf_cache.rs +274 -0
  111. package/native/shared/src/shadows.rs +1036 -0
  112. package/native/shared/src/staging.rs +102 -0
  113. package/native/shared/src/string_header.rs +266 -0
  114. package/native/shared/src/text_renderer.rs +502 -0
  115. package/native/shared/src/textures.rs +197 -0
  116. package/native/tvos/Cargo.lock +1693 -0
  117. package/native/tvos/Cargo.toml +36 -0
  118. package/native/tvos/metal-patched/Cargo.toml +178 -0
  119. package/native/tvos/metal-patched/LICENSE-APACHE +201 -0
  120. package/native/tvos/metal-patched/LICENSE-MIT +25 -0
  121. package/native/tvos/metal-patched/src/acceleration_structure.rs +667 -0
  122. package/native/tvos/metal-patched/src/acceleration_structure_pass.rs +108 -0
  123. package/native/tvos/metal-patched/src/argument.rs +366 -0
  124. package/native/tvos/metal-patched/src/blitpass.rs +102 -0
  125. package/native/tvos/metal-patched/src/buffer.rs +71 -0
  126. package/native/tvos/metal-patched/src/capturedescriptor.rs +76 -0
  127. package/native/tvos/metal-patched/src/capturemanager.rs +113 -0
  128. package/native/tvos/metal-patched/src/commandbuffer.rs +192 -0
  129. package/native/tvos/metal-patched/src/commandqueue.rs +44 -0
  130. package/native/tvos/metal-patched/src/computepass.rs +107 -0
  131. package/native/tvos/metal-patched/src/constants.rs +152 -0
  132. package/native/tvos/metal-patched/src/counters.rs +119 -0
  133. package/native/tvos/metal-patched/src/depthstencil.rs +190 -0
  134. package/native/tvos/metal-patched/src/device.rs +2134 -0
  135. package/native/tvos/metal-patched/src/drawable.rs +39 -0
  136. package/native/tvos/metal-patched/src/encoder.rs +2041 -0
  137. package/native/tvos/metal-patched/src/heap.rs +281 -0
  138. package/native/tvos/metal-patched/src/indirect_encoder.rs +344 -0
  139. package/native/tvos/metal-patched/src/lib.rs +657 -0
  140. package/native/tvos/metal-patched/src/library.rs +902 -0
  141. package/native/tvos/metal-patched/src/mps.rs +575 -0
  142. package/native/tvos/metal-patched/src/pipeline/compute.rs +475 -0
  143. package/native/tvos/metal-patched/src/pipeline/mod.rs +71 -0
  144. package/native/tvos/metal-patched/src/pipeline/render.rs +762 -0
  145. package/native/tvos/metal-patched/src/renderpass.rs +443 -0
  146. package/native/tvos/metal-patched/src/resource.rs +182 -0
  147. package/native/tvos/metal-patched/src/sampler.rs +165 -0
  148. package/native/tvos/metal-patched/src/sync.rs +178 -0
  149. package/native/tvos/metal-patched/src/texture.rs +352 -0
  150. package/native/tvos/metal-patched/src/types.rs +90 -0
  151. package/native/tvos/metal-patched/src/vertexdescriptor.rs +250 -0
  152. package/native/tvos/src/audio_backend.rs +197 -0
  153. package/native/tvos/src/lib.rs +1891 -0
  154. package/native/visionos/Cargo.lock +1693 -0
  155. package/native/visionos/Cargo.toml +40 -0
  156. package/native/visionos/src/audio_backend.rs +197 -0
  157. package/native/visionos/src/lib.rs +1887 -0
  158. package/native/watchos/Cargo.lock +16 -0
  159. package/native/watchos/Cargo.toml +19 -0
  160. package/native/watchos/shaders/bloom_postfx.metal +99 -0
  161. package/native/watchos/src/BloomWatchApp.swift +1267 -0
  162. package/native/watchos/src/BloomWatchAudio.swift +179 -0
  163. package/native/watchos/src/audio.rs +55 -0
  164. package/native/watchos/src/draw_list.rs +229 -0
  165. package/native/watchos/src/ffi_stubs.rs +915 -0
  166. package/native/watchos/src/ffi_stubs_manual.rs +35 -0
  167. package/native/watchos/src/lib.rs +1124 -0
  168. package/native/watchos/src/models.rs +746 -0
  169. package/native/watchos/src/postfx.rs +95 -0
  170. package/native/watchos/src/scene.rs +534 -0
  171. package/native/watchos/src/textures.rs +184 -0
  172. package/native/web/Cargo.lock +1657 -0
  173. package/native/web/Cargo.toml +43 -0
  174. package/native/web/bloom_glue.js +695 -0
  175. package/native/web/build.sh +131 -0
  176. package/native/web/index.html +35 -0
  177. package/native/web/jolt_bridge.js +1519 -0
  178. package/native/web/src/input_ffi.rs +286 -0
  179. package/native/web/src/lib.rs +1796 -0
  180. package/native/web/src/material_ffi.rs +710 -0
  181. package/native/web/src/parity_ffi.rs +343 -0
  182. package/native/web/src/physics_ffi.rs +643 -0
  183. package/native/web/src/ragdoll_ffi.rs +250 -0
  184. package/native/web/src/render_settings.rs +98 -0
  185. package/native/windows/Cargo.lock +1815 -0
  186. package/native/windows/Cargo.toml +68 -0
  187. package/native/windows/src/lib.rs +1486 -0
  188. package/package.json +4279 -0
  189. package/src/audio/index.ts +315 -0
  190. package/src/core/colors.ts +63 -0
  191. package/src/core/index.ts +1206 -0
  192. package/src/core/keys.ts +63 -0
  193. package/src/core/types.ts +104 -0
  194. package/src/index.ts +171 -0
  195. package/src/math/index.ts +516 -0
  196. package/src/mobile/index.ts +294 -0
  197. package/src/models/index.ts +1258 -0
  198. package/src/physics/index.ts +1134 -0
  199. package/src/scene/index.ts +698 -0
  200. package/src/shapes/index.ts +120 -0
  201. package/src/text/index.ts +48 -0
  202. package/src/textures/index.ts +187 -0
  203. package/src/vfx/index.ts +191 -0
  204. package/src/world/index.ts +24 -0
  205. package/src/world/loader.ts +423 -0
  206. package/src/world/prefab.ts +217 -0
  207. package/src/world/render.ts +172 -0
  208. package/src/world/saver.ts +108 -0
  209. package/src/world/serialize.ts +301 -0
  210. package/src/world/terrain.ts +355 -0
  211. package/src/world/types.ts +160 -0
  212. package/src/world/validate.ts +319 -0
  213. package/src/world/version.ts +114 -0
@@ -0,0 +1,577 @@
1
+ //! PT-1 — progressive path-trace megakernel dispatch
2
+ //! (docs/pt/PT-1-progressive-megakernel.md). Replaces the lit opaque
3
+ //! scene colour in hdr_rt when path tracing is active; sky pixels are
4
+ //! left untouched so the raster sky/clouds survive, and translucency
5
+ //! still composites on top afterwards.
6
+
7
+ use super::*;
8
+
9
+ impl Renderer {
10
+ pub(super) fn record_pt_pass(
11
+ &mut self,
12
+ encoder: &mut wgpu::CommandEncoder,
13
+ profiler: &mut crate::profiler::Profiler,
14
+ surf_w: u32,
15
+ surf_h: u32,
16
+ ) {
17
+ if !self.pt_active() {
18
+ // Leaving PT (or never entering it) invalidates history so
19
+ // re-enabling starts a fresh accumulation, not a stale one.
20
+ self.pt_accum_count = 0;
21
+ self.pt_wrote_frame = false;
22
+ return;
23
+ }
24
+ // Same readiness gate as the HW probe trace: first frames before
25
+ // any geometry is committed have no TLAS / instance data yet.
26
+ if self.pt_pipeline.is_none()
27
+ || self.tlas.is_none()
28
+ || self.tlas_instance_data_buffer.is_none()
29
+ || self.pt_geo_vertex_buffer.is_none()
30
+ || self.pt_geo_index_buffer.is_none()
31
+ {
32
+ self.pt_accum_count = 0;
33
+ self.pt_wrote_frame = false;
34
+ return;
35
+ }
36
+ // PT-2 — a grown texture store means the baked view array is
37
+ // stale; rebuild so new textures become visible to hit shading.
38
+ if self.pt_texture_arrays_enabled && self.pt_bg_texture_count != self.textures.len() {
39
+ self.pt_tex_bg = None;
40
+ }
41
+
42
+ // ---- accumulation validity ----
43
+ // Any camera motion beyond epsilon restarts progressive
44
+ // accumulation (mode 1). Mode 2 (realtime) ignores the reset —
45
+ // its EMA is designed to absorb motion — but tracking prev_vp
46
+ // costs nothing and keeps one code path.
47
+ //
48
+ // Compared UNJITTERED: current_vp_matrix carries the TAA Halton
49
+ // nudge (~1e-3 in the proj Z-coupling slots), which would read
50
+ // as motion every frame and pin the accumulator at 1 sample.
51
+ // The jittered inv_vp still goes to the kernel — primary rays
52
+ // must match the jittered G-buffer depth, and accumulating
53
+ // across jitters is free anti-aliasing.
54
+ let vp_unjittered = mat4_multiply(
55
+ self.current_proj_matrix_unjittered,
56
+ self.current_view_matrix,
57
+ );
58
+ let mut moved = false;
59
+ for r in 0..4 {
60
+ for c in 0..4 {
61
+ if (vp_unjittered[r][c] - self.pt_prev_vp[r][c]).abs() > 1e-5 {
62
+ moved = true;
63
+ }
64
+ }
65
+ }
66
+ if moved && self.pt_mode == 1 {
67
+ self.pt_accum_count = 0;
68
+ }
69
+ // PT-3 — the uniform needs LAST frame's VP for history
70
+ // reprojection; stash it before the tracker is overwritten.
71
+ let prev_vp_for_reproject = self.pt_prev_vp;
72
+ self.pt_prev_vp = vp_unjittered;
73
+ // Geometry changed under the accumulated image (door opened,
74
+ // enemy died) → PROGRESSIVE history is a lie, restart. Realtime
75
+ // must NOT reset here: tlas_version bumps on every node
76
+ // transform — during gameplay that is every single frame, which
77
+ // silently pinned the SVGF history at 1 sample (found via the
78
+ // debug-20 history-length view; the frozen-seed era masked it).
79
+ // Mode 2's per-tap depth validation already rejects exactly the
80
+ // texels whose surface actually changed.
81
+ let mut tlas_reset = false;
82
+ if self.tlas_built_version != self.pt_last_tlas_version {
83
+ self.pt_last_tlas_version = self.tlas_built_version;
84
+ if self.pt_mode == 1 {
85
+ self.pt_accum_count = 0;
86
+ tlas_reset = true;
87
+ }
88
+ }
89
+ // Progressive mode + camera in motion OR scene churn: the raster
90
+ // frame stays on screen (kernel write threshold) and any sample
91
+ // traced now is discarded by next frame's reset — skip the
92
+ // dispatch entirely. During combat the TLAS bumps every frame
93
+ // (enemy transforms), so without the tlas_reset arm progressive
94
+ // paid the full-res trace cost while displaying raster.
95
+ if self.pt_mode == 1 && (moved || tlas_reset) {
96
+ self.pt_wrote_frame = false;
97
+ return;
98
+ }
99
+
100
+ // ---- trace grid ----
101
+ // Realtime mode traces at half resolution (4x fewer rays) and
102
+ // joint-bilaterally upsamples in the final à-trous pass; the
103
+ // 2x2 sample phase rotates per frame so the temporal EMA
104
+ // integrates full-res coverage over 4 frames. Progressive mode
105
+ // stays full-res.
106
+ // The realtime trace grid is capped at ~0.5 Mpx (960x540) so
107
+ // raising the raster render scale sharpens the image without
108
+ // multiplying the ray budget — the upsampler handles arbitrary
109
+ // trace-to-full ratios. Progressive stays uncapped: quality is
110
+ // its entire point.
111
+ let (trace_w, trace_h) = if self.pt_mode >= 2 {
112
+ (surf_w.div_ceil(2).min(960), surf_h.div_ceil(2).min(540))
113
+ } else {
114
+ (surf_w, surf_h)
115
+ };
116
+ // Phase pinned to (0,0): rotating it makes each trace texel
117
+ // sample a different full-res pixel every frame, and on
118
+ // depth-chaotic surfaces (grass) the history validation then
119
+ // rejects almost every frame — texels never accumulate past
120
+ // 1 spp and read as white speckle. A consistent owner pixel
121
+ // keeps history valid; the upsample covers the other three.
122
+ let _phase = [0u32, 0u32];
123
+
124
+ // ---- accumulation buffers (vec4<f32> per pixel, ping-pong) ----
125
+ // Sized to the TRACE grid; a mode switch changes the size and
126
+ // recreates (which also resets accumulation — correct, the two
127
+ // modes' buffer contents are not interchangeable).
128
+ let needed = (trace_w as u64) * (trace_h as u64) * 16;
129
+ let recreate = match &self.pt_accum_buffers[0] {
130
+ Some(b) => b.size() != needed,
131
+ None => true,
132
+ };
133
+ if recreate {
134
+ for (i, slot) in self.pt_accum_buffers.iter_mut().enumerate() {
135
+ *slot = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
136
+ label: Some(if i == 0 { "pt_accum_a" } else { "pt_accum_b" }),
137
+ size: needed,
138
+ // COPY_SRC: the debug-16 numeric readback copies a
139
+ // window of this buffer to a staging buffer.
140
+ usage: wgpu::BufferUsages::STORAGE
141
+ | wgpu::BufferUsages::COPY_DST
142
+ | wgpu::BufferUsages::COPY_SRC,
143
+ mapped_at_creation: false,
144
+ }));
145
+ }
146
+ // SVGF moments side-channel (mu1, mu2, history length, raw
147
+ // depth), ping-pong with the accum pair. wgpu zero-inits,
148
+ // and pt_accum_count = 0 marks the whole history invalid.
149
+ for (i, slot) in self.pt_moments_buffers.iter_mut().enumerate() {
150
+ *slot = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
151
+ label: Some(if i == 0 { "pt_moments_a" } else { "pt_moments_b" }),
152
+ size: needed,
153
+ usage: wgpu::BufferUsages::STORAGE | wgpu::BufferUsages::COPY_DST,
154
+ mapped_at_creation: false,
155
+ }));
156
+ }
157
+ // PT-4 — ReSTIR reservoirs (light idx, W, M, target pdf).
158
+ // Zero-init M = 0 marks every reservoir empty.
159
+ for (i, slot) in self.pt_resv_buffers.iter_mut().enumerate() {
160
+ *slot = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
161
+ label: Some(if i == 0 { "pt_resv_a" } else { "pt_resv_b" }),
162
+ size: needed,
163
+ usage: wgpu::BufferUsages::STORAGE | wgpu::BufferUsages::COPY_DST,
164
+ mapped_at_creation: false,
165
+ }));
166
+ }
167
+ // COPY_SRC: the first à-trous iteration's output is copied
168
+ // back over the accum buffer as next frame's colour history
169
+ // (SVGF feeds back the once-filtered signal).
170
+ self.pt_atrous_scratch = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
171
+ label: Some("pt_atrous_scratch"),
172
+ size: needed,
173
+ usage: wgpu::BufferUsages::STORAGE | wgpu::BufferUsages::COPY_SRC,
174
+ mapped_at_creation: false,
175
+ }));
176
+ self.pt_atrous_scratch2 = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
177
+ label: Some("pt_atrous_scratch2"),
178
+ size: needed,
179
+ usage: wgpu::BufferUsages::STORAGE,
180
+ mapped_at_creation: false,
181
+ }));
182
+ self.pt_bg = [None, None];
183
+ self.pt_atrous_bgs = [[None, None, None, None, None, None], [None, None, None, None, None, None]];
184
+ self.pt_accum_count = 0;
185
+ self.pt_accum_idx = 0;
186
+ }
187
+
188
+ // ---- uniforms ----
189
+ // Sun / sky derivation matches record_ssgi_passes exactly so PT
190
+ // brightness lines up with the raster + GI frame it replaces.
191
+ let ld = self.lighting_uniforms.light_dir;
192
+ let sun_inv_len = 1.0 / (ld[0] * ld[0] + ld[1] * ld[1] + ld[2] * ld[2]).sqrt().max(1e-4);
193
+ let sun_intensity = ld[3].max(0.0);
194
+ let lc = self.lighting_uniforms.light_color;
195
+ let amb = self.lighting_uniforms.ambient;
196
+ let sky_intensity = amb[3].max(0.0);
197
+
198
+ let light_count = (self.lighting_uniforms.point_light_count[0] as usize).min(16);
199
+ let mut lights = [[0.0f32; 4]; 32];
200
+ for i in 0..light_count {
201
+ let pl = &self.lighting_uniforms.point_lights[i];
202
+ lights[i * 2] = pl.position; // xyz + range
203
+ lights[i * 2 + 1] = pl.color; // rgb + intensity
204
+ }
205
+
206
+ let max_bounces = if self.pt_mode == 2 { 2.0 } else { 8.0 };
207
+ let cam = self.current_camera_pos;
208
+ // current_inv_vp_matrix is stored transposed relative to what
209
+ // WGSL's `M * v` needs (the composed VP inherits mat4_multiply's
210
+ // convention; its inverse lands transposed). Upload the
211
+ // transpose so the kernel's unprojection is the real inverse —
212
+ // without this every ray collapses to one degenerate bundle and
213
+ // the whole path trace silently hits garbage (found via numeric
214
+ // readback; see docs/pt/PT-2 notes).
215
+ let m = &self.current_inv_vp_matrix;
216
+ let inv_vp_t = [
217
+ [m[0][0], m[1][0], m[2][0], m[3][0]],
218
+ [m[0][1], m[1][1], m[2][1], m[3][1]],
219
+ [m[0][2], m[1][2], m[2][2], m[3][2]],
220
+ [m[0][3], m[1][3], m[2][3], m[3][3]],
221
+ ];
222
+ // The reprojection VP uploads RAW — the opposite of inv_vp. The
223
+ // two matrix conventions coexist: mat4_invert outputs land
224
+ // transposed relative to WGSL's M*v (hence inv_vp's transpose
225
+ // above), while mat4_multiply products are already in M*v
226
+ // layout — the shadow cascade VPs upload raw for the same
227
+ // reason. Transposing this one collapsed every reprojection
228
+ // into a ~40-texel band at screen centre (debug-23 dump), so
229
+ // history never matched under camera motion — invisible in the
230
+ // frozen-seed era, which is why it survived since PT-3 M1.
231
+ let prev_vp_t = prev_vp_for_reproject;
232
+ let params = PtParamsCpu {
233
+ inv_vp: inv_vp_t,
234
+ prev_vp: prev_vp_t,
235
+ cam_pos: [cam[0], cam[1], cam[2], 0.0],
236
+ sun_dir: [
237
+ -ld[0] * sun_inv_len,
238
+ -ld[1] * sun_inv_len,
239
+ -ld[2] * sun_inv_len,
240
+ 0.0,
241
+ ],
242
+ sun_color: [
243
+ lc[0] * sun_intensity,
244
+ lc[1] * sun_intensity,
245
+ lc[2] * sun_intensity,
246
+ 0.0,
247
+ ],
248
+ sky_color: [
249
+ amb[0] * sky_intensity,
250
+ amb[1] * sky_intensity,
251
+ amb[2] * sky_intensity,
252
+ 0.0,
253
+ ],
254
+ // size.z: PT's OWN frame counter, not taa_frame_index —
255
+ // the TAA index freezes when TAA is disabled (settings,
256
+ // headless tests), which froze the sample sequence and
257
+ // silently stopped progressive accumulation from ever
258
+ // converging (found by the pt_progressive golden).
259
+ size: [trace_w, trace_h, self.pt_frame_index, self.pt_accum_count],
260
+ cfg: [
261
+ self.pt_mode as f32,
262
+ max_bounces,
263
+ light_count as f32,
264
+ self.pt_debug,
265
+ ],
266
+ // ext.z: hybrid sun — realtime mode samples the raster
267
+ // shadow cascades instead of tracing sun rays (crisp
268
+ // noise-free direct shadows). Progressive keeps traced sun
269
+ // for reference-quality penumbra.
270
+ ext: [
271
+ surf_w,
272
+ surf_h,
273
+ if self.pt_mode >= 2 && self.shadow_map.enabled { 1 } else { 0 },
274
+ // PT-4 experimental flag (BLOOM_PT_RESTIR=1), realtime only.
275
+ if self.pt_restir && self.pt_mode >= 2 { 1 } else { 0 },
276
+ ],
277
+ // RAW upload, unlike inv_vp: the shadow VPs are consumed as
278
+ // M*v by every existing WGSL user (scene shader, WSRC
279
+ // bake), so they are already stored in WGSL column layout.
280
+ // Verified empirically via debug 18 — transposing them
281
+ // black-shadows the whole frame.
282
+ shadow_vps: self.shadow_map.light_vps,
283
+ lights,
284
+ };
285
+ self.queue.write_buffer(&self.pt_uniform_buffer, 0, bytemuck::bytes_of(&params));
286
+
287
+ // ---- bind groups (lazy; nulled on resize / TLAS or instance
288
+ // buffer recreation). Two ping-pong variants: bg[i] reads accum
289
+ // buffer i (binding 8) and writes buffer 1-i (binding 13).
290
+ for i in 0..2 {
291
+ if self.pt_bg[i].is_some() {
292
+ continue;
293
+ }
294
+ let tlas = self.tlas.as_ref().unwrap();
295
+ let entries = vec![
296
+ wgpu::BindGroupEntry { binding: 0, resource: self.pt_uniform_buffer.as_entire_binding() },
297
+ wgpu::BindGroupEntry { binding: 1, resource: tlas.as_binding() },
298
+ wgpu::BindGroupEntry { binding: 2, resource: self.tlas_instance_data_buffer.as_ref().unwrap().as_entire_binding() },
299
+ wgpu::BindGroupEntry { binding: 3, resource: wgpu::BindingResource::TextureView(&self.depth_view) },
300
+ wgpu::BindGroupEntry { binding: 4, resource: wgpu::BindingResource::TextureView(&self.albedo_rt_view) },
301
+ wgpu::BindGroupEntry { binding: 5, resource: wgpu::BindingResource::TextureView(&self.material_rt_view) },
302
+ // Raw albedo atlas, NOT the pre-lit radiance atlas the
303
+ // GI probe trace uses — PT computes its own lighting at
304
+ // hits; radiance would double-count.
305
+ wgpu::BindGroupEntry { binding: 6, resource: wgpu::BindingResource::TextureView(&self.mesh_card_atlas_view) },
306
+ wgpu::BindGroupEntry { binding: 7, resource: wgpu::BindingResource::Sampler(&self.mesh_card_atlas_sampler) },
307
+ wgpu::BindGroupEntry { binding: 8, resource: self.pt_accum_buffers[i].as_ref().unwrap().as_entire_binding() },
308
+ wgpu::BindGroupEntry { binding: 9, resource: wgpu::BindingResource::TextureView(&self.hdr_rt_view) },
309
+ wgpu::BindGroupEntry { binding: 10, resource: self.pt_geo_vertex_buffer.as_ref().unwrap().as_entire_binding() },
310
+ wgpu::BindGroupEntry { binding: 11, resource: self.pt_geo_index_buffer.as_ref().unwrap().as_entire_binding() },
311
+ wgpu::BindGroupEntry { binding: 13, resource: self.pt_accum_buffers[1 - i].as_ref().unwrap().as_entire_binding() },
312
+ wgpu::BindGroupEntry { binding: 14, resource: wgpu::BindingResource::TextureView(&self.shadow_map.depth_views[0]) },
313
+ wgpu::BindGroupEntry { binding: 15, resource: wgpu::BindingResource::TextureView(&self.shadow_map.depth_views[1]) },
314
+ wgpu::BindGroupEntry { binding: 16, resource: wgpu::BindingResource::TextureView(&self.shadow_map.depth_views[2]) },
315
+ wgpu::BindGroupEntry { binding: 17, resource: wgpu::BindingResource::Sampler(&self.shadow_map.sampler) },
316
+ // SVGF moments: read prev (paired with accum read
317
+ // side), write out (paired with the write side).
318
+ wgpu::BindGroupEntry { binding: 18, resource: self.pt_moments_buffers[i].as_ref().unwrap().as_entire_binding() },
319
+ wgpu::BindGroupEntry { binding: 19, resource: self.pt_moments_buffers[1 - i].as_ref().unwrap().as_entire_binding() },
320
+ // PT-4 ReSTIR reservoirs, same ping-pong pairing.
321
+ wgpu::BindGroupEntry { binding: 20, resource: self.pt_resv_buffers[i].as_ref().unwrap().as_entire_binding() },
322
+ wgpu::BindGroupEntry { binding: 21, resource: self.pt_resv_buffers[1 - i].as_ref().unwrap().as_entire_binding() },
323
+ // PT-7 — velocity MRT (written by hdr_scene, which
324
+ // runs before the PT node every frame).
325
+ wgpu::BindGroupEntry { binding: 22, resource: wgpu::BindingResource::TextureView(&self.velocity_rt_view) },
326
+ ];
327
+ self.pt_bg[i] = Some(self.device.create_bind_group(&wgpu::BindGroupDescriptor {
328
+ label: Some("pt_bg"),
329
+ layout: self.pt_layout.as_ref().unwrap(),
330
+ entries: &entries,
331
+ }));
332
+ }
333
+ // PT-2 — group 1: the texture binding array. Real store views
334
+ // first, white (slot 0) padding to the fixed layout count. The
335
+ // bind group holds refs, so the temporary views live with it.
336
+ if self.pt_texture_arrays_enabled && self.pt_tex_bg.is_none() {
337
+ let n = self.textures.len().min(PT_MAX_TEXTURES);
338
+ let tex_views: Vec<wgpu::TextureView> = (0..n.max(1))
339
+ .map(|i| self.textures[i.min(self.textures.len() - 1)]
340
+ .create_view(&wgpu::TextureViewDescriptor::default()))
341
+ .collect();
342
+ let tex_view_refs: Vec<&wgpu::TextureView> = (0..PT_MAX_TEXTURES)
343
+ .map(|i| &tex_views[if i < n { i } else { 0 }])
344
+ .collect();
345
+ self.pt_tex_bg = Some(self.device.create_bind_group(&wgpu::BindGroupDescriptor {
346
+ label: Some("pt_tex_bg"),
347
+ layout: self.pt_tex_layout.as_ref().unwrap(),
348
+ entries: &[wgpu::BindGroupEntry {
349
+ binding: 0,
350
+ resource: wgpu::BindingResource::TextureViewArray(&tex_view_refs),
351
+ }],
352
+ }));
353
+ self.pt_bg_texture_count = self.textures.len();
354
+ }
355
+
356
+ // ---- dispatch ----
357
+ {
358
+ let ts = profiler.compute_pass_timestamp_writes("pt_pass");
359
+ let mut pass = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor {
360
+ label: Some("pt_pass"),
361
+ timestamp_writes: ts,
362
+ });
363
+ pass.set_pipeline(self.pt_pipeline.as_ref().unwrap());
364
+ pass.set_bind_group(0, self.pt_bg[self.pt_accum_idx].as_ref().unwrap(), &[]);
365
+ if self.pt_texture_arrays_enabled {
366
+ pass.set_bind_group(1, self.pt_tex_bg.as_ref().unwrap(), &[]);
367
+ }
368
+ pass.dispatch_workgroups((trace_w + 7) / 8, (trace_h + 7) / 8, 1);
369
+ }
370
+ // This frame wrote into buffers[1 - idx]; it becomes next
371
+ // frame's read side.
372
+ let written_idx = 1 - self.pt_accum_idx;
373
+ self.pt_accum_idx = written_idx;
374
+
375
+ // ---- PT-3b: SVGF wavelet filter (realtime mode only) ----
376
+ // Six variance-guided à-trous iterations on the trace grid
377
+ // (steps 1/2/4/8/16/1), then the full-res upsample+modulate pass.
378
+ // After iteration 1 the once-filtered signal is copied back
379
+ // over the accum buffer: SVGF feeds the first wavelet output
380
+ // into next frame's colour history (moments stay raw). This is
381
+ // what makes the temporal loop stable at 1 spp — raw history
382
+ // carries every spike forward, once-filtered history does not.
383
+ // Progressive mode converges on its own and writes hdr
384
+ // directly from the kernel.
385
+ if self.pt_mode >= 2
386
+ && self.pt_debug == 0.0
387
+ && self.pt_atrous_mid_pipeline.is_some()
388
+ && self.pt_atrous_scratch.is_some()
389
+ {
390
+ // p.y = 1.0 flags the FIRST iteration: it may substitute a
391
+ // spatial variance estimate where the history is young.
392
+ for (i, step) in [1.0f32, 2.0, 4.0, 8.0, 16.0, 1.0].iter().enumerate() {
393
+ let first = if i == 0 { 1.0f32 } else { 0.0 };
394
+ let p = [
395
+ [*step, first, trace_w as f32, trace_h as f32],
396
+ [surf_w as f32, surf_h as f32, 0.0, 0.0],
397
+ ];
398
+ self.queue.write_buffer(&self.pt_atrous_params_bufs[i], 0, bytemuck::bytes_of(&p));
399
+ }
400
+
401
+ if self.pt_atrous_bgs[written_idx][0].is_none() {
402
+ let scratch = self.pt_atrous_scratch.as_ref().unwrap();
403
+ let scratch2 = self.pt_atrous_scratch2.as_ref().unwrap();
404
+ let accum_w = self.pt_accum_buffers[written_idx].as_ref().unwrap();
405
+ let moments_w = self.pt_moments_buffers[written_idx].as_ref().unwrap();
406
+ // Stage src → dst chain: accum→s1, then the scratches
407
+ // ping-pong; the final upsample reads the last-written
408
+ // scratch. cs_final never writes dst; it gets whichever
409
+ // scratch is not its src (RO+RW of one buffer in a
410
+ // single group fails validation).
411
+ let chain: [(&wgpu::Buffer, &wgpu::Buffer); 6] = [
412
+ (accum_w, scratch),
413
+ (scratch, scratch2),
414
+ (scratch2, scratch),
415
+ (scratch, scratch2),
416
+ (scratch2, scratch),
417
+ (scratch, scratch2),
418
+ ];
419
+ for (i, (src, dst)) in chain.iter().enumerate() {
420
+ self.pt_atrous_bgs[written_idx][i] = Some(self.device.create_bind_group(&wgpu::BindGroupDescriptor {
421
+ label: Some("pt_atrous_bg"),
422
+ layout: self.pt_atrous_layout.as_ref().unwrap(),
423
+ entries: &[
424
+ wgpu::BindGroupEntry { binding: 0, resource: self.pt_atrous_params_bufs[i].as_entire_binding() },
425
+ wgpu::BindGroupEntry { binding: 1, resource: src.as_entire_binding() },
426
+ wgpu::BindGroupEntry { binding: 2, resource: dst.as_entire_binding() },
427
+ wgpu::BindGroupEntry { binding: 3, resource: wgpu::BindingResource::TextureView(&self.hdr_rt_view) },
428
+ wgpu::BindGroupEntry { binding: 4, resource: wgpu::BindingResource::TextureView(&self.depth_view) },
429
+ wgpu::BindGroupEntry { binding: 5, resource: wgpu::BindingResource::TextureView(&self.albedo_rt_view) },
430
+ wgpu::BindGroupEntry { binding: 6, resource: moments_w.as_entire_binding() },
431
+ ],
432
+ }));
433
+ }
434
+ }
435
+
436
+ {
437
+ let ts = profiler.compute_pass_timestamp_writes("pt_atrous");
438
+ let mut pass = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor {
439
+ label: Some("pt_atrous"),
440
+ timestamp_writes: ts,
441
+ });
442
+ pass.set_pipeline(self.pt_atrous_mid_pipeline.as_ref().unwrap());
443
+ pass.set_bind_group(0, self.pt_atrous_bgs[written_idx][0].as_ref().unwrap(), &[]);
444
+ pass.dispatch_workgroups((trace_w + 7) / 8, (trace_h + 7) / 8, 1);
445
+ }
446
+ // History feedback: the pass split makes the copy legal
447
+ // (buffer copies cannot live inside a compute pass).
448
+ encoder.copy_buffer_to_buffer(
449
+ self.pt_atrous_scratch.as_ref().unwrap(),
450
+ 0,
451
+ self.pt_accum_buffers[written_idx].as_ref().unwrap(),
452
+ 0,
453
+ needed,
454
+ );
455
+ {
456
+ let ts = profiler.compute_pass_timestamp_writes("pt_atrous2");
457
+ let mut pass = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor {
458
+ label: Some("pt_atrous2"),
459
+ timestamp_writes: ts,
460
+ });
461
+ pass.set_pipeline(self.pt_atrous_mid_pipeline.as_ref().unwrap());
462
+ for i in 1..5 {
463
+ pass.set_bind_group(0, self.pt_atrous_bgs[written_idx][i].as_ref().unwrap(), &[]);
464
+ pass.dispatch_workgroups((trace_w + 7) / 8, (trace_h + 7) / 8, 1);
465
+ }
466
+ pass.set_pipeline(self.pt_atrous_final_pipeline.as_ref().unwrap());
467
+ pass.set_bind_group(0, self.pt_atrous_bgs[written_idx][5].as_ref().unwrap(), &[]);
468
+ pass.dispatch_workgroups((surf_w + 7) / 8, (surf_h + 7) / 8, 1);
469
+ }
470
+ }
471
+ // Mirrors the kernel's write threshold: mode 1 leaves the raster
472
+ // frame on screen until 8 samples exist (u.size.w carried the
473
+ // pre-increment count), so SSGI/SSR must keep running for those
474
+ // frames — the gates downstream check pt_owns_frame().
475
+ self.pt_wrote_frame = self.pt_mode >= 2 || self.pt_accum_count >= 8;
476
+ self.pt_accum_count = self.pt_accum_count.saturating_add(1);
477
+ self.pt_frame_index = self.pt_frame_index.wrapping_add(1);
478
+
479
+ // ---- debug 16: numeric readback of traced intersections ----
480
+ // Copies a window of the accum buffer (center of frame) into a
481
+ // staging buffer each frame; the previous frame's copy is mapped
482
+ // (blocking) and dumped to pt_trace_dump.txt once.
483
+ if (self.pt_debug == 16.0
484
+ || self.pt_debug == 17.0
485
+ || self.pt_debug == 19.0
486
+ || self.pt_debug == 22.0
487
+ || self.pt_debug == 23.0)
488
+ && self.pt_accum_count > 30
489
+ && !self.pt_dump_written
490
+ {
491
+ // Offsets in TRACE-grid units — the accum buffers are
492
+ // half-res in realtime mode.
493
+ let dump_pixels: u64 = (trace_w as u64).min(4096);
494
+ let row = (trace_h / 2) as u64;
495
+ let offset = row * trace_w as u64 * 16;
496
+ if self.pt_readback_buffer.is_none() {
497
+ self.pt_readback_buffer = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
498
+ label: Some("pt_readback"),
499
+ size: dump_pixels * 16,
500
+ usage: wgpu::BufferUsages::MAP_READ | wgpu::BufferUsages::COPY_DST,
501
+ mapped_at_creation: false,
502
+ }));
503
+ encoder.copy_buffer_to_buffer(
504
+ // written_idx == pt_accum_idx here (already flipped):
505
+ // the buffer this frame's dispatch wrote.
506
+ self.pt_accum_buffers[self.pt_accum_idx].as_ref().unwrap(),
507
+ offset,
508
+ self.pt_readback_buffer.as_ref().unwrap(),
509
+ 0,
510
+ dump_pixels * 16,
511
+ );
512
+ } else {
513
+ // Previous frame's copy has been submitted; map it now.
514
+ let buf = self.pt_readback_buffer.as_ref().unwrap();
515
+ let slice = buf.slice(..);
516
+ slice.map_async(wgpu::MapMode::Read, |_| {});
517
+ let _ = self.device.poll(wgpu::PollType::Wait { submission_index: None, timeout: None });
518
+ let data = slice.get_mapped_range();
519
+ let vals: &[[f32; 4]] = bytemuck::cast_slice(&data);
520
+ let mut out = String::new();
521
+ out.push_str(&format!(
522
+ "middle row, {} pixels, mode {}\n",
523
+ vals.len(),
524
+ self.pt_debug
525
+ ));
526
+ // Every 64th pixel across the full row. Field meaning:
527
+ // 16 = t / id / prim / kind; 17 = p0.xyz / raw depth.
528
+ for (i, v) in vals.iter().enumerate().step_by(64) {
529
+ out.push_str(&format!(
530
+ "col {i}: {:.4} {:.4} {:.4} {:.6}\n",
531
+ v[0], v[1], v[2], v[3]
532
+ ));
533
+ }
534
+ // Also dump the CPU-side uniform inputs for comparison,
535
+ // plus the unprojection computed in BOTH multiply
536
+ // conventions. Whichever matches the GPU dump is what
537
+ // the shader effectively computed; the other (if sane)
538
+ // is the fix.
539
+ let ndc = [0.0f32, 0.0, 0.998647, 1.0];
540
+ let m = &self.current_inv_vp_matrix;
541
+ let mut h_col = [0.0f32; 4]; // h_i = sum_c m[c][i] * ndc[c]
542
+ let mut h_row = [0.0f32; 4]; // h_i = sum_c m[i][c] * ndc[c]
543
+ for i in 0..4 {
544
+ for c in 0..4 {
545
+ h_col[i] += m[c][i] * ndc[c];
546
+ h_row[i] += m[i][c] * ndc[c];
547
+ }
548
+ }
549
+ out.push_str(&format!(
550
+ "cpu cam_pos = {:?}\n\
551
+ unproject as columns: h={:?} p={:?}\n\
552
+ unproject transposed: h={:?} p={:?}\n",
553
+ self.current_camera_pos,
554
+ h_col,
555
+ [h_col[0] / h_col[3], h_col[1] / h_col[3], h_col[2] / h_col[3]],
556
+ h_row,
557
+ [h_row[0] / h_row[3], h_row[1] / h_row[3], h_row[2] / h_row[3]],
558
+ ));
559
+ // Distinct instance-id count over the whole row.
560
+ let mut ids: Vec<i64> = vals
561
+ .iter()
562
+ .filter(|v| v[3] != 0.0)
563
+ .map(|v| v[1] as i64)
564
+ .collect();
565
+ ids.sort_unstable();
566
+ ids.dedup();
567
+ let misses = vals.iter().filter(|v| v[3] == 0.0).count();
568
+ out.push_str(&format!("distinct hit ids: {:?}\n", ids));
569
+ out.push_str(&format!("misses: {}\n", misses));
570
+ drop(data);
571
+ buf.unmap();
572
+ let _ = std::fs::write("pt_trace_dump.txt", out);
573
+ self.pt_dump_written = true;
574
+ }
575
+ }
576
+ }
577
+ }