@bornengine/engine 0.4.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +231 -0
- package/native/android/Cargo.lock +1848 -0
- package/native/android/Cargo.toml +24 -0
- package/native/android/src/lib.rs +702 -0
- package/native/ios/Cargo.lock +1690 -0
- package/native/ios/Cargo.toml +32 -0
- package/native/ios/src/lib.rs +1267 -0
- package/native/linux/Cargo.lock +3279 -0
- package/native/linux/Cargo.toml +29 -0
- package/native/linux/src/lib.rs +1331 -0
- package/native/macos/Cargo.lock +3310 -0
- package/native/macos/Cargo.toml +46 -0
- package/native/macos/src/lib.rs +1302 -0
- package/native/shared/Cargo.lock +1899 -0
- package/native/shared/Cargo.toml +62 -0
- package/native/shared/assets/default_font.ttf +0 -0
- package/native/shared/build.rs +270 -0
- package/native/shared/shaders/common/clouds.wgsl +122 -0
- package/native/shared/shaders/common/fog.wgsl +16 -0
- package/native/shared/shaders/common/foliage_wind.wgsl +98 -0
- package/native/shared/shaders/common/imposter.wgsl +112 -0
- package/native/shared/shaders/common/pbr.wgsl +186 -0
- package/native/shared/shaders/common/shadows.wgsl +186 -0
- package/native/shared/shaders/common/sky.wgsl +8 -0
- package/native/shared/shaders/common/tonemap.wgsl +25 -0
- package/native/shared/shaders/impulse_field.wgsl +57 -0
- package/native/shared/shaders/material_abi.wgsl +383 -0
- package/native/shared/shaders/materials/test_minimal.wgsl +42 -0
- package/native/shared/src/anim_mixer.rs +61 -0
- package/native/shared/src/attach.rs +263 -0
- package/native/shared/src/audio/decode.rs +123 -0
- package/native/shared/src/audio/mod.rs +863 -0
- package/native/shared/src/audio/render.rs +892 -0
- package/native/shared/src/audio/spsc.rs +156 -0
- package/native/shared/src/audio/stream.rs +226 -0
- package/native/shared/src/custom_shaders.rs +104 -0
- package/native/shared/src/decals.rs +245 -0
- package/native/shared/src/drs.rs +211 -0
- package/native/shared/src/engine.rs +261 -0
- package/native/shared/src/ffi.rs +116 -0
- package/native/shared/src/ffi_core/assets.rs +388 -0
- package/native/shared/src/ffi_core/audio_ffi.rs +184 -0
- package/native/shared/src/ffi_core/draw.rs +334 -0
- package/native/shared/src/ffi_core/game_loop.rs +577 -0
- package/native/shared/src/ffi_core/input.rs +234 -0
- package/native/shared/src/ffi_core/mod.rs +127 -0
- package/native/shared/src/ffi_core/models.rs +1154 -0
- package/native/shared/src/ffi_core/ragdoll_ffi.rs +261 -0
- package/native/shared/src/ffi_core/scene.rs +626 -0
- package/native/shared/src/ffi_core/vfx.rs +212 -0
- package/native/shared/src/ffi_core/visual.rs +691 -0
- package/native/shared/src/frame_callbacks.rs +122 -0
- package/native/shared/src/geometry.rs +236 -0
- package/native/shared/src/handles.rs +182 -0
- package/native/shared/src/input.rs +448 -0
- package/native/shared/src/jolt_sys.rs +822 -0
- package/native/shared/src/lib.rs +55 -0
- package/native/shared/src/models.rs +1093 -0
- package/native/shared/src/models_gltf.rs +1280 -0
- package/native/shared/src/particles.rs +391 -0
- package/native/shared/src/physics_jolt.rs +1908 -0
- package/native/shared/src/picking.rs +298 -0
- package/native/shared/src/postfx.rs +345 -0
- package/native/shared/src/profiler.rs +492 -0
- package/native/shared/src/ragdoll.rs +474 -0
- package/native/shared/src/renderer/atmosphere_lut.rs +573 -0
- package/native/shared/src/renderer/brdf_lut.rs +154 -0
- package/native/shared/src/renderer/draw2d.rs +143 -0
- package/native/shared/src/renderer/formats.rs +822 -0
- package/native/shared/src/renderer/froxel.rs +421 -0
- package/native/shared/src/renderer/gi_bake.rs +653 -0
- package/native/shared/src/renderer/graph.rs +462 -0
- package/native/shared/src/renderer/hiz.rs +269 -0
- package/native/shared/src/renderer/hot_reload.rs +390 -0
- package/native/shared/src/renderer/impulse_field.rs +456 -0
- package/native/shared/src/renderer/lighting.rs +154 -0
- package/native/shared/src/renderer/material_instancing.rs +171 -0
- package/native/shared/src/renderer/material_pipeline.rs +700 -0
- package/native/shared/src/renderer/material_system.rs +1996 -0
- package/native/shared/src/renderer/material_system_tests.rs +601 -0
- package/native/shared/src/renderer/material_system_wasm.rs +41 -0
- package/native/shared/src/renderer/mod.rs +12556 -0
- package/native/shared/src/renderer/model_draw.rs +641 -0
- package/native/shared/src/renderer/occlusion.rs +429 -0
- package/native/shared/src/renderer/planar_pass.rs +593 -0
- package/native/shared/src/renderer/planar_reflection.rs +499 -0
- package/native/shared/src/renderer/post_pass.rs +249 -0
- package/native/shared/src/renderer/postfx_chain.rs +728 -0
- package/native/shared/src/renderer/pt_pass.rs +577 -0
- package/native/shared/src/renderer/scene_pass.rs +607 -0
- package/native/shared/src/renderer/shader_include.rs +205 -0
- package/native/shared/src/renderer/shader_library.rs +135 -0
- package/native/shared/src/renderer/shaders/ao.rs +570 -0
- package/native/shared/src/renderer/shaders/core.rs +1243 -0
- package/native/shared/src/renderer/shaders/env.rs +907 -0
- package/native/shared/src/renderer/shaders/gi.rs +810 -0
- package/native/shared/src/renderer/shaders/mod.rs +19 -0
- package/native/shared/src/renderer/shaders/post.rs +1558 -0
- package/native/shared/src/renderer/shaders/pt.rs +1859 -0
- package/native/shared/src/renderer/shaders/ssgi.rs +1586 -0
- package/native/shared/src/renderer/shadow_pass.rs +731 -0
- package/native/shared/src/renderer/ssgi_pass.rs +392 -0
- package/native/shared/src/renderer/ssr_pass.rs +188 -0
- package/native/shared/src/renderer/texture_store.rs +473 -0
- package/native/shared/src/renderer/transient.rs +591 -0
- package/native/shared/src/renderer/types.rs +941 -0
- package/native/shared/src/renderer/util.rs +152 -0
- package/native/shared/src/scene.rs +1362 -0
- package/native/shared/src/sdf_cache.rs +274 -0
- package/native/shared/src/shadows.rs +1036 -0
- package/native/shared/src/staging.rs +102 -0
- package/native/shared/src/string_header.rs +266 -0
- package/native/shared/src/text_renderer.rs +502 -0
- package/native/shared/src/textures.rs +197 -0
- package/native/tvos/Cargo.lock +1693 -0
- package/native/tvos/Cargo.toml +36 -0
- package/native/tvos/metal-patched/Cargo.toml +178 -0
- package/native/tvos/metal-patched/LICENSE-APACHE +201 -0
- package/native/tvos/metal-patched/LICENSE-MIT +25 -0
- package/native/tvos/metal-patched/src/acceleration_structure.rs +667 -0
- package/native/tvos/metal-patched/src/acceleration_structure_pass.rs +108 -0
- package/native/tvos/metal-patched/src/argument.rs +366 -0
- package/native/tvos/metal-patched/src/blitpass.rs +102 -0
- package/native/tvos/metal-patched/src/buffer.rs +71 -0
- package/native/tvos/metal-patched/src/capturedescriptor.rs +76 -0
- package/native/tvos/metal-patched/src/capturemanager.rs +113 -0
- package/native/tvos/metal-patched/src/commandbuffer.rs +192 -0
- package/native/tvos/metal-patched/src/commandqueue.rs +44 -0
- package/native/tvos/metal-patched/src/computepass.rs +107 -0
- package/native/tvos/metal-patched/src/constants.rs +152 -0
- package/native/tvos/metal-patched/src/counters.rs +119 -0
- package/native/tvos/metal-patched/src/depthstencil.rs +190 -0
- package/native/tvos/metal-patched/src/device.rs +2134 -0
- package/native/tvos/metal-patched/src/drawable.rs +39 -0
- package/native/tvos/metal-patched/src/encoder.rs +2041 -0
- package/native/tvos/metal-patched/src/heap.rs +281 -0
- package/native/tvos/metal-patched/src/indirect_encoder.rs +344 -0
- package/native/tvos/metal-patched/src/lib.rs +657 -0
- package/native/tvos/metal-patched/src/library.rs +902 -0
- package/native/tvos/metal-patched/src/mps.rs +575 -0
- package/native/tvos/metal-patched/src/pipeline/compute.rs +475 -0
- package/native/tvos/metal-patched/src/pipeline/mod.rs +71 -0
- package/native/tvos/metal-patched/src/pipeline/render.rs +762 -0
- package/native/tvos/metal-patched/src/renderpass.rs +443 -0
- package/native/tvos/metal-patched/src/resource.rs +182 -0
- package/native/tvos/metal-patched/src/sampler.rs +165 -0
- package/native/tvos/metal-patched/src/sync.rs +178 -0
- package/native/tvos/metal-patched/src/texture.rs +352 -0
- package/native/tvos/metal-patched/src/types.rs +90 -0
- package/native/tvos/metal-patched/src/vertexdescriptor.rs +250 -0
- package/native/tvos/src/audio_backend.rs +197 -0
- package/native/tvos/src/lib.rs +1891 -0
- package/native/visionos/Cargo.lock +1693 -0
- package/native/visionos/Cargo.toml +40 -0
- package/native/visionos/src/audio_backend.rs +197 -0
- package/native/visionos/src/lib.rs +1887 -0
- package/native/watchos/Cargo.lock +16 -0
- package/native/watchos/Cargo.toml +19 -0
- package/native/watchos/shaders/bloom_postfx.metal +99 -0
- package/native/watchos/src/BloomWatchApp.swift +1267 -0
- package/native/watchos/src/BloomWatchAudio.swift +179 -0
- package/native/watchos/src/audio.rs +55 -0
- package/native/watchos/src/draw_list.rs +229 -0
- package/native/watchos/src/ffi_stubs.rs +915 -0
- package/native/watchos/src/ffi_stubs_manual.rs +35 -0
- package/native/watchos/src/lib.rs +1124 -0
- package/native/watchos/src/models.rs +746 -0
- package/native/watchos/src/postfx.rs +95 -0
- package/native/watchos/src/scene.rs +534 -0
- package/native/watchos/src/textures.rs +184 -0
- package/native/web/Cargo.lock +1657 -0
- package/native/web/Cargo.toml +43 -0
- package/native/web/bloom_glue.js +695 -0
- package/native/web/build.sh +131 -0
- package/native/web/index.html +35 -0
- package/native/web/jolt_bridge.js +1519 -0
- package/native/web/src/input_ffi.rs +286 -0
- package/native/web/src/lib.rs +1796 -0
- package/native/web/src/material_ffi.rs +710 -0
- package/native/web/src/parity_ffi.rs +343 -0
- package/native/web/src/physics_ffi.rs +643 -0
- package/native/web/src/ragdoll_ffi.rs +250 -0
- package/native/web/src/render_settings.rs +98 -0
- package/native/windows/Cargo.lock +1815 -0
- package/native/windows/Cargo.toml +68 -0
- package/native/windows/src/lib.rs +1486 -0
- package/package.json +4279 -0
- package/src/audio/index.ts +315 -0
- package/src/core/colors.ts +63 -0
- package/src/core/index.ts +1206 -0
- package/src/core/keys.ts +63 -0
- package/src/core/types.ts +104 -0
- package/src/index.ts +171 -0
- package/src/math/index.ts +516 -0
- package/src/mobile/index.ts +294 -0
- package/src/models/index.ts +1258 -0
- package/src/physics/index.ts +1134 -0
- package/src/scene/index.ts +698 -0
- package/src/shapes/index.ts +120 -0
- package/src/text/index.ts +48 -0
- package/src/textures/index.ts +187 -0
- package/src/vfx/index.ts +191 -0
- package/src/world/index.ts +24 -0
- package/src/world/loader.ts +423 -0
- package/src/world/prefab.ts +217 -0
- package/src/world/render.ts +172 -0
- package/src/world/saver.ts +108 -0
- package/src/world/serialize.ts +301 -0
- package/src/world/terrain.ts +355 -0
- package/src/world/types.ts +160 -0
- package/src/world/validate.ts +319 -0
- package/src/world/version.ts +114 -0
|
@@ -0,0 +1,577 @@
|
|
|
1
|
+
//! PT-1 — progressive path-trace megakernel dispatch
|
|
2
|
+
//! (docs/pt/PT-1-progressive-megakernel.md). Replaces the lit opaque
|
|
3
|
+
//! scene colour in hdr_rt when path tracing is active; sky pixels are
|
|
4
|
+
//! left untouched so the raster sky/clouds survive, and translucency
|
|
5
|
+
//! still composites on top afterwards.
|
|
6
|
+
|
|
7
|
+
use super::*;
|
|
8
|
+
|
|
9
|
+
impl Renderer {
|
|
10
|
+
pub(super) fn record_pt_pass(
|
|
11
|
+
&mut self,
|
|
12
|
+
encoder: &mut wgpu::CommandEncoder,
|
|
13
|
+
profiler: &mut crate::profiler::Profiler,
|
|
14
|
+
surf_w: u32,
|
|
15
|
+
surf_h: u32,
|
|
16
|
+
) {
|
|
17
|
+
if !self.pt_active() {
|
|
18
|
+
// Leaving PT (or never entering it) invalidates history so
|
|
19
|
+
// re-enabling starts a fresh accumulation, not a stale one.
|
|
20
|
+
self.pt_accum_count = 0;
|
|
21
|
+
self.pt_wrote_frame = false;
|
|
22
|
+
return;
|
|
23
|
+
}
|
|
24
|
+
// Same readiness gate as the HW probe trace: first frames before
|
|
25
|
+
// any geometry is committed have no TLAS / instance data yet.
|
|
26
|
+
if self.pt_pipeline.is_none()
|
|
27
|
+
|| self.tlas.is_none()
|
|
28
|
+
|| self.tlas_instance_data_buffer.is_none()
|
|
29
|
+
|| self.pt_geo_vertex_buffer.is_none()
|
|
30
|
+
|| self.pt_geo_index_buffer.is_none()
|
|
31
|
+
{
|
|
32
|
+
self.pt_accum_count = 0;
|
|
33
|
+
self.pt_wrote_frame = false;
|
|
34
|
+
return;
|
|
35
|
+
}
|
|
36
|
+
// PT-2 — a grown texture store means the baked view array is
|
|
37
|
+
// stale; rebuild so new textures become visible to hit shading.
|
|
38
|
+
if self.pt_texture_arrays_enabled && self.pt_bg_texture_count != self.textures.len() {
|
|
39
|
+
self.pt_tex_bg = None;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// ---- accumulation validity ----
|
|
43
|
+
// Any camera motion beyond epsilon restarts progressive
|
|
44
|
+
// accumulation (mode 1). Mode 2 (realtime) ignores the reset —
|
|
45
|
+
// its EMA is designed to absorb motion — but tracking prev_vp
|
|
46
|
+
// costs nothing and keeps one code path.
|
|
47
|
+
//
|
|
48
|
+
// Compared UNJITTERED: current_vp_matrix carries the TAA Halton
|
|
49
|
+
// nudge (~1e-3 in the proj Z-coupling slots), which would read
|
|
50
|
+
// as motion every frame and pin the accumulator at 1 sample.
|
|
51
|
+
// The jittered inv_vp still goes to the kernel — primary rays
|
|
52
|
+
// must match the jittered G-buffer depth, and accumulating
|
|
53
|
+
// across jitters is free anti-aliasing.
|
|
54
|
+
let vp_unjittered = mat4_multiply(
|
|
55
|
+
self.current_proj_matrix_unjittered,
|
|
56
|
+
self.current_view_matrix,
|
|
57
|
+
);
|
|
58
|
+
let mut moved = false;
|
|
59
|
+
for r in 0..4 {
|
|
60
|
+
for c in 0..4 {
|
|
61
|
+
if (vp_unjittered[r][c] - self.pt_prev_vp[r][c]).abs() > 1e-5 {
|
|
62
|
+
moved = true;
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
if moved && self.pt_mode == 1 {
|
|
67
|
+
self.pt_accum_count = 0;
|
|
68
|
+
}
|
|
69
|
+
// PT-3 — the uniform needs LAST frame's VP for history
|
|
70
|
+
// reprojection; stash it before the tracker is overwritten.
|
|
71
|
+
let prev_vp_for_reproject = self.pt_prev_vp;
|
|
72
|
+
self.pt_prev_vp = vp_unjittered;
|
|
73
|
+
// Geometry changed under the accumulated image (door opened,
|
|
74
|
+
// enemy died) → PROGRESSIVE history is a lie, restart. Realtime
|
|
75
|
+
// must NOT reset here: tlas_version bumps on every node
|
|
76
|
+
// transform — during gameplay that is every single frame, which
|
|
77
|
+
// silently pinned the SVGF history at 1 sample (found via the
|
|
78
|
+
// debug-20 history-length view; the frozen-seed era masked it).
|
|
79
|
+
// Mode 2's per-tap depth validation already rejects exactly the
|
|
80
|
+
// texels whose surface actually changed.
|
|
81
|
+
let mut tlas_reset = false;
|
|
82
|
+
if self.tlas_built_version != self.pt_last_tlas_version {
|
|
83
|
+
self.pt_last_tlas_version = self.tlas_built_version;
|
|
84
|
+
if self.pt_mode == 1 {
|
|
85
|
+
self.pt_accum_count = 0;
|
|
86
|
+
tlas_reset = true;
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
// Progressive mode + camera in motion OR scene churn: the raster
|
|
90
|
+
// frame stays on screen (kernel write threshold) and any sample
|
|
91
|
+
// traced now is discarded by next frame's reset — skip the
|
|
92
|
+
// dispatch entirely. During combat the TLAS bumps every frame
|
|
93
|
+
// (enemy transforms), so without the tlas_reset arm progressive
|
|
94
|
+
// paid the full-res trace cost while displaying raster.
|
|
95
|
+
if self.pt_mode == 1 && (moved || tlas_reset) {
|
|
96
|
+
self.pt_wrote_frame = false;
|
|
97
|
+
return;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
// ---- trace grid ----
|
|
101
|
+
// Realtime mode traces at half resolution (4x fewer rays) and
|
|
102
|
+
// joint-bilaterally upsamples in the final à-trous pass; the
|
|
103
|
+
// 2x2 sample phase rotates per frame so the temporal EMA
|
|
104
|
+
// integrates full-res coverage over 4 frames. Progressive mode
|
|
105
|
+
// stays full-res.
|
|
106
|
+
// The realtime trace grid is capped at ~0.5 Mpx (960x540) so
|
|
107
|
+
// raising the raster render scale sharpens the image without
|
|
108
|
+
// multiplying the ray budget — the upsampler handles arbitrary
|
|
109
|
+
// trace-to-full ratios. Progressive stays uncapped: quality is
|
|
110
|
+
// its entire point.
|
|
111
|
+
let (trace_w, trace_h) = if self.pt_mode >= 2 {
|
|
112
|
+
(surf_w.div_ceil(2).min(960), surf_h.div_ceil(2).min(540))
|
|
113
|
+
} else {
|
|
114
|
+
(surf_w, surf_h)
|
|
115
|
+
};
|
|
116
|
+
// Phase pinned to (0,0): rotating it makes each trace texel
|
|
117
|
+
// sample a different full-res pixel every frame, and on
|
|
118
|
+
// depth-chaotic surfaces (grass) the history validation then
|
|
119
|
+
// rejects almost every frame — texels never accumulate past
|
|
120
|
+
// 1 spp and read as white speckle. A consistent owner pixel
|
|
121
|
+
// keeps history valid; the upsample covers the other three.
|
|
122
|
+
let _phase = [0u32, 0u32];
|
|
123
|
+
|
|
124
|
+
// ---- accumulation buffers (vec4<f32> per pixel, ping-pong) ----
|
|
125
|
+
// Sized to the TRACE grid; a mode switch changes the size and
|
|
126
|
+
// recreates (which also resets accumulation — correct, the two
|
|
127
|
+
// modes' buffer contents are not interchangeable).
|
|
128
|
+
let needed = (trace_w as u64) * (trace_h as u64) * 16;
|
|
129
|
+
let recreate = match &self.pt_accum_buffers[0] {
|
|
130
|
+
Some(b) => b.size() != needed,
|
|
131
|
+
None => true,
|
|
132
|
+
};
|
|
133
|
+
if recreate {
|
|
134
|
+
for (i, slot) in self.pt_accum_buffers.iter_mut().enumerate() {
|
|
135
|
+
*slot = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
|
|
136
|
+
label: Some(if i == 0 { "pt_accum_a" } else { "pt_accum_b" }),
|
|
137
|
+
size: needed,
|
|
138
|
+
// COPY_SRC: the debug-16 numeric readback copies a
|
|
139
|
+
// window of this buffer to a staging buffer.
|
|
140
|
+
usage: wgpu::BufferUsages::STORAGE
|
|
141
|
+
| wgpu::BufferUsages::COPY_DST
|
|
142
|
+
| wgpu::BufferUsages::COPY_SRC,
|
|
143
|
+
mapped_at_creation: false,
|
|
144
|
+
}));
|
|
145
|
+
}
|
|
146
|
+
// SVGF moments side-channel (mu1, mu2, history length, raw
|
|
147
|
+
// depth), ping-pong with the accum pair. wgpu zero-inits,
|
|
148
|
+
// and pt_accum_count = 0 marks the whole history invalid.
|
|
149
|
+
for (i, slot) in self.pt_moments_buffers.iter_mut().enumerate() {
|
|
150
|
+
*slot = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
|
|
151
|
+
label: Some(if i == 0 { "pt_moments_a" } else { "pt_moments_b" }),
|
|
152
|
+
size: needed,
|
|
153
|
+
usage: wgpu::BufferUsages::STORAGE | wgpu::BufferUsages::COPY_DST,
|
|
154
|
+
mapped_at_creation: false,
|
|
155
|
+
}));
|
|
156
|
+
}
|
|
157
|
+
// PT-4 — ReSTIR reservoirs (light idx, W, M, target pdf).
|
|
158
|
+
// Zero-init M = 0 marks every reservoir empty.
|
|
159
|
+
for (i, slot) in self.pt_resv_buffers.iter_mut().enumerate() {
|
|
160
|
+
*slot = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
|
|
161
|
+
label: Some(if i == 0 { "pt_resv_a" } else { "pt_resv_b" }),
|
|
162
|
+
size: needed,
|
|
163
|
+
usage: wgpu::BufferUsages::STORAGE | wgpu::BufferUsages::COPY_DST,
|
|
164
|
+
mapped_at_creation: false,
|
|
165
|
+
}));
|
|
166
|
+
}
|
|
167
|
+
// COPY_SRC: the first à-trous iteration's output is copied
|
|
168
|
+
// back over the accum buffer as next frame's colour history
|
|
169
|
+
// (SVGF feeds back the once-filtered signal).
|
|
170
|
+
self.pt_atrous_scratch = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
|
|
171
|
+
label: Some("pt_atrous_scratch"),
|
|
172
|
+
size: needed,
|
|
173
|
+
usage: wgpu::BufferUsages::STORAGE | wgpu::BufferUsages::COPY_SRC,
|
|
174
|
+
mapped_at_creation: false,
|
|
175
|
+
}));
|
|
176
|
+
self.pt_atrous_scratch2 = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
|
|
177
|
+
label: Some("pt_atrous_scratch2"),
|
|
178
|
+
size: needed,
|
|
179
|
+
usage: wgpu::BufferUsages::STORAGE,
|
|
180
|
+
mapped_at_creation: false,
|
|
181
|
+
}));
|
|
182
|
+
self.pt_bg = [None, None];
|
|
183
|
+
self.pt_atrous_bgs = [[None, None, None, None, None, None], [None, None, None, None, None, None]];
|
|
184
|
+
self.pt_accum_count = 0;
|
|
185
|
+
self.pt_accum_idx = 0;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// ---- uniforms ----
|
|
189
|
+
// Sun / sky derivation matches record_ssgi_passes exactly so PT
|
|
190
|
+
// brightness lines up with the raster + GI frame it replaces.
|
|
191
|
+
let ld = self.lighting_uniforms.light_dir;
|
|
192
|
+
let sun_inv_len = 1.0 / (ld[0] * ld[0] + ld[1] * ld[1] + ld[2] * ld[2]).sqrt().max(1e-4);
|
|
193
|
+
let sun_intensity = ld[3].max(0.0);
|
|
194
|
+
let lc = self.lighting_uniforms.light_color;
|
|
195
|
+
let amb = self.lighting_uniforms.ambient;
|
|
196
|
+
let sky_intensity = amb[3].max(0.0);
|
|
197
|
+
|
|
198
|
+
let light_count = (self.lighting_uniforms.point_light_count[0] as usize).min(16);
|
|
199
|
+
let mut lights = [[0.0f32; 4]; 32];
|
|
200
|
+
for i in 0..light_count {
|
|
201
|
+
let pl = &self.lighting_uniforms.point_lights[i];
|
|
202
|
+
lights[i * 2] = pl.position; // xyz + range
|
|
203
|
+
lights[i * 2 + 1] = pl.color; // rgb + intensity
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
let max_bounces = if self.pt_mode == 2 { 2.0 } else { 8.0 };
|
|
207
|
+
let cam = self.current_camera_pos;
|
|
208
|
+
// current_inv_vp_matrix is stored transposed relative to what
|
|
209
|
+
// WGSL's `M * v` needs (the composed VP inherits mat4_multiply's
|
|
210
|
+
// convention; its inverse lands transposed). Upload the
|
|
211
|
+
// transpose so the kernel's unprojection is the real inverse —
|
|
212
|
+
// without this every ray collapses to one degenerate bundle and
|
|
213
|
+
// the whole path trace silently hits garbage (found via numeric
|
|
214
|
+
// readback; see docs/pt/PT-2 notes).
|
|
215
|
+
let m = &self.current_inv_vp_matrix;
|
|
216
|
+
let inv_vp_t = [
|
|
217
|
+
[m[0][0], m[1][0], m[2][0], m[3][0]],
|
|
218
|
+
[m[0][1], m[1][1], m[2][1], m[3][1]],
|
|
219
|
+
[m[0][2], m[1][2], m[2][2], m[3][2]],
|
|
220
|
+
[m[0][3], m[1][3], m[2][3], m[3][3]],
|
|
221
|
+
];
|
|
222
|
+
// The reprojection VP uploads RAW — the opposite of inv_vp. The
|
|
223
|
+
// two matrix conventions coexist: mat4_invert outputs land
|
|
224
|
+
// transposed relative to WGSL's M*v (hence inv_vp's transpose
|
|
225
|
+
// above), while mat4_multiply products are already in M*v
|
|
226
|
+
// layout — the shadow cascade VPs upload raw for the same
|
|
227
|
+
// reason. Transposing this one collapsed every reprojection
|
|
228
|
+
// into a ~40-texel band at screen centre (debug-23 dump), so
|
|
229
|
+
// history never matched under camera motion — invisible in the
|
|
230
|
+
// frozen-seed era, which is why it survived since PT-3 M1.
|
|
231
|
+
let prev_vp_t = prev_vp_for_reproject;
|
|
232
|
+
let params = PtParamsCpu {
|
|
233
|
+
inv_vp: inv_vp_t,
|
|
234
|
+
prev_vp: prev_vp_t,
|
|
235
|
+
cam_pos: [cam[0], cam[1], cam[2], 0.0],
|
|
236
|
+
sun_dir: [
|
|
237
|
+
-ld[0] * sun_inv_len,
|
|
238
|
+
-ld[1] * sun_inv_len,
|
|
239
|
+
-ld[2] * sun_inv_len,
|
|
240
|
+
0.0,
|
|
241
|
+
],
|
|
242
|
+
sun_color: [
|
|
243
|
+
lc[0] * sun_intensity,
|
|
244
|
+
lc[1] * sun_intensity,
|
|
245
|
+
lc[2] * sun_intensity,
|
|
246
|
+
0.0,
|
|
247
|
+
],
|
|
248
|
+
sky_color: [
|
|
249
|
+
amb[0] * sky_intensity,
|
|
250
|
+
amb[1] * sky_intensity,
|
|
251
|
+
amb[2] * sky_intensity,
|
|
252
|
+
0.0,
|
|
253
|
+
],
|
|
254
|
+
// size.z: PT's OWN frame counter, not taa_frame_index —
|
|
255
|
+
// the TAA index freezes when TAA is disabled (settings,
|
|
256
|
+
// headless tests), which froze the sample sequence and
|
|
257
|
+
// silently stopped progressive accumulation from ever
|
|
258
|
+
// converging (found by the pt_progressive golden).
|
|
259
|
+
size: [trace_w, trace_h, self.pt_frame_index, self.pt_accum_count],
|
|
260
|
+
cfg: [
|
|
261
|
+
self.pt_mode as f32,
|
|
262
|
+
max_bounces,
|
|
263
|
+
light_count as f32,
|
|
264
|
+
self.pt_debug,
|
|
265
|
+
],
|
|
266
|
+
// ext.z: hybrid sun — realtime mode samples the raster
|
|
267
|
+
// shadow cascades instead of tracing sun rays (crisp
|
|
268
|
+
// noise-free direct shadows). Progressive keeps traced sun
|
|
269
|
+
// for reference-quality penumbra.
|
|
270
|
+
ext: [
|
|
271
|
+
surf_w,
|
|
272
|
+
surf_h,
|
|
273
|
+
if self.pt_mode >= 2 && self.shadow_map.enabled { 1 } else { 0 },
|
|
274
|
+
// PT-4 experimental flag (BLOOM_PT_RESTIR=1), realtime only.
|
|
275
|
+
if self.pt_restir && self.pt_mode >= 2 { 1 } else { 0 },
|
|
276
|
+
],
|
|
277
|
+
// RAW upload, unlike inv_vp: the shadow VPs are consumed as
|
|
278
|
+
// M*v by every existing WGSL user (scene shader, WSRC
|
|
279
|
+
// bake), so they are already stored in WGSL column layout.
|
|
280
|
+
// Verified empirically via debug 18 — transposing them
|
|
281
|
+
// black-shadows the whole frame.
|
|
282
|
+
shadow_vps: self.shadow_map.light_vps,
|
|
283
|
+
lights,
|
|
284
|
+
};
|
|
285
|
+
self.queue.write_buffer(&self.pt_uniform_buffer, 0, bytemuck::bytes_of(¶ms));
|
|
286
|
+
|
|
287
|
+
// ---- bind groups (lazy; nulled on resize / TLAS or instance
|
|
288
|
+
// buffer recreation). Two ping-pong variants: bg[i] reads accum
|
|
289
|
+
// buffer i (binding 8) and writes buffer 1-i (binding 13).
|
|
290
|
+
for i in 0..2 {
|
|
291
|
+
if self.pt_bg[i].is_some() {
|
|
292
|
+
continue;
|
|
293
|
+
}
|
|
294
|
+
let tlas = self.tlas.as_ref().unwrap();
|
|
295
|
+
let entries = vec![
|
|
296
|
+
wgpu::BindGroupEntry { binding: 0, resource: self.pt_uniform_buffer.as_entire_binding() },
|
|
297
|
+
wgpu::BindGroupEntry { binding: 1, resource: tlas.as_binding() },
|
|
298
|
+
wgpu::BindGroupEntry { binding: 2, resource: self.tlas_instance_data_buffer.as_ref().unwrap().as_entire_binding() },
|
|
299
|
+
wgpu::BindGroupEntry { binding: 3, resource: wgpu::BindingResource::TextureView(&self.depth_view) },
|
|
300
|
+
wgpu::BindGroupEntry { binding: 4, resource: wgpu::BindingResource::TextureView(&self.albedo_rt_view) },
|
|
301
|
+
wgpu::BindGroupEntry { binding: 5, resource: wgpu::BindingResource::TextureView(&self.material_rt_view) },
|
|
302
|
+
// Raw albedo atlas, NOT the pre-lit radiance atlas the
|
|
303
|
+
// GI probe trace uses — PT computes its own lighting at
|
|
304
|
+
// hits; radiance would double-count.
|
|
305
|
+
wgpu::BindGroupEntry { binding: 6, resource: wgpu::BindingResource::TextureView(&self.mesh_card_atlas_view) },
|
|
306
|
+
wgpu::BindGroupEntry { binding: 7, resource: wgpu::BindingResource::Sampler(&self.mesh_card_atlas_sampler) },
|
|
307
|
+
wgpu::BindGroupEntry { binding: 8, resource: self.pt_accum_buffers[i].as_ref().unwrap().as_entire_binding() },
|
|
308
|
+
wgpu::BindGroupEntry { binding: 9, resource: wgpu::BindingResource::TextureView(&self.hdr_rt_view) },
|
|
309
|
+
wgpu::BindGroupEntry { binding: 10, resource: self.pt_geo_vertex_buffer.as_ref().unwrap().as_entire_binding() },
|
|
310
|
+
wgpu::BindGroupEntry { binding: 11, resource: self.pt_geo_index_buffer.as_ref().unwrap().as_entire_binding() },
|
|
311
|
+
wgpu::BindGroupEntry { binding: 13, resource: self.pt_accum_buffers[1 - i].as_ref().unwrap().as_entire_binding() },
|
|
312
|
+
wgpu::BindGroupEntry { binding: 14, resource: wgpu::BindingResource::TextureView(&self.shadow_map.depth_views[0]) },
|
|
313
|
+
wgpu::BindGroupEntry { binding: 15, resource: wgpu::BindingResource::TextureView(&self.shadow_map.depth_views[1]) },
|
|
314
|
+
wgpu::BindGroupEntry { binding: 16, resource: wgpu::BindingResource::TextureView(&self.shadow_map.depth_views[2]) },
|
|
315
|
+
wgpu::BindGroupEntry { binding: 17, resource: wgpu::BindingResource::Sampler(&self.shadow_map.sampler) },
|
|
316
|
+
// SVGF moments: read prev (paired with accum read
|
|
317
|
+
// side), write out (paired with the write side).
|
|
318
|
+
wgpu::BindGroupEntry { binding: 18, resource: self.pt_moments_buffers[i].as_ref().unwrap().as_entire_binding() },
|
|
319
|
+
wgpu::BindGroupEntry { binding: 19, resource: self.pt_moments_buffers[1 - i].as_ref().unwrap().as_entire_binding() },
|
|
320
|
+
// PT-4 ReSTIR reservoirs, same ping-pong pairing.
|
|
321
|
+
wgpu::BindGroupEntry { binding: 20, resource: self.pt_resv_buffers[i].as_ref().unwrap().as_entire_binding() },
|
|
322
|
+
wgpu::BindGroupEntry { binding: 21, resource: self.pt_resv_buffers[1 - i].as_ref().unwrap().as_entire_binding() },
|
|
323
|
+
// PT-7 — velocity MRT (written by hdr_scene, which
|
|
324
|
+
// runs before the PT node every frame).
|
|
325
|
+
wgpu::BindGroupEntry { binding: 22, resource: wgpu::BindingResource::TextureView(&self.velocity_rt_view) },
|
|
326
|
+
];
|
|
327
|
+
self.pt_bg[i] = Some(self.device.create_bind_group(&wgpu::BindGroupDescriptor {
|
|
328
|
+
label: Some("pt_bg"),
|
|
329
|
+
layout: self.pt_layout.as_ref().unwrap(),
|
|
330
|
+
entries: &entries,
|
|
331
|
+
}));
|
|
332
|
+
}
|
|
333
|
+
// PT-2 — group 1: the texture binding array. Real store views
|
|
334
|
+
// first, white (slot 0) padding to the fixed layout count. The
|
|
335
|
+
// bind group holds refs, so the temporary views live with it.
|
|
336
|
+
if self.pt_texture_arrays_enabled && self.pt_tex_bg.is_none() {
|
|
337
|
+
let n = self.textures.len().min(PT_MAX_TEXTURES);
|
|
338
|
+
let tex_views: Vec<wgpu::TextureView> = (0..n.max(1))
|
|
339
|
+
.map(|i| self.textures[i.min(self.textures.len() - 1)]
|
|
340
|
+
.create_view(&wgpu::TextureViewDescriptor::default()))
|
|
341
|
+
.collect();
|
|
342
|
+
let tex_view_refs: Vec<&wgpu::TextureView> = (0..PT_MAX_TEXTURES)
|
|
343
|
+
.map(|i| &tex_views[if i < n { i } else { 0 }])
|
|
344
|
+
.collect();
|
|
345
|
+
self.pt_tex_bg = Some(self.device.create_bind_group(&wgpu::BindGroupDescriptor {
|
|
346
|
+
label: Some("pt_tex_bg"),
|
|
347
|
+
layout: self.pt_tex_layout.as_ref().unwrap(),
|
|
348
|
+
entries: &[wgpu::BindGroupEntry {
|
|
349
|
+
binding: 0,
|
|
350
|
+
resource: wgpu::BindingResource::TextureViewArray(&tex_view_refs),
|
|
351
|
+
}],
|
|
352
|
+
}));
|
|
353
|
+
self.pt_bg_texture_count = self.textures.len();
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
// ---- dispatch ----
|
|
357
|
+
{
|
|
358
|
+
let ts = profiler.compute_pass_timestamp_writes("pt_pass");
|
|
359
|
+
let mut pass = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor {
|
|
360
|
+
label: Some("pt_pass"),
|
|
361
|
+
timestamp_writes: ts,
|
|
362
|
+
});
|
|
363
|
+
pass.set_pipeline(self.pt_pipeline.as_ref().unwrap());
|
|
364
|
+
pass.set_bind_group(0, self.pt_bg[self.pt_accum_idx].as_ref().unwrap(), &[]);
|
|
365
|
+
if self.pt_texture_arrays_enabled {
|
|
366
|
+
pass.set_bind_group(1, self.pt_tex_bg.as_ref().unwrap(), &[]);
|
|
367
|
+
}
|
|
368
|
+
pass.dispatch_workgroups((trace_w + 7) / 8, (trace_h + 7) / 8, 1);
|
|
369
|
+
}
|
|
370
|
+
// This frame wrote into buffers[1 - idx]; it becomes next
|
|
371
|
+
// frame's read side.
|
|
372
|
+
let written_idx = 1 - self.pt_accum_idx;
|
|
373
|
+
self.pt_accum_idx = written_idx;
|
|
374
|
+
|
|
375
|
+
// ---- PT-3b: SVGF wavelet filter (realtime mode only) ----
|
|
376
|
+
// Six variance-guided à-trous iterations on the trace grid
|
|
377
|
+
// (steps 1/2/4/8/16/1), then the full-res upsample+modulate pass.
|
|
378
|
+
// After iteration 1 the once-filtered signal is copied back
|
|
379
|
+
// over the accum buffer: SVGF feeds the first wavelet output
|
|
380
|
+
// into next frame's colour history (moments stay raw). This is
|
|
381
|
+
// what makes the temporal loop stable at 1 spp — raw history
|
|
382
|
+
// carries every spike forward, once-filtered history does not.
|
|
383
|
+
// Progressive mode converges on its own and writes hdr
|
|
384
|
+
// directly from the kernel.
|
|
385
|
+
if self.pt_mode >= 2
|
|
386
|
+
&& self.pt_debug == 0.0
|
|
387
|
+
&& self.pt_atrous_mid_pipeline.is_some()
|
|
388
|
+
&& self.pt_atrous_scratch.is_some()
|
|
389
|
+
{
|
|
390
|
+
// p.y = 1.0 flags the FIRST iteration: it may substitute a
|
|
391
|
+
// spatial variance estimate where the history is young.
|
|
392
|
+
for (i, step) in [1.0f32, 2.0, 4.0, 8.0, 16.0, 1.0].iter().enumerate() {
|
|
393
|
+
let first = if i == 0 { 1.0f32 } else { 0.0 };
|
|
394
|
+
let p = [
|
|
395
|
+
[*step, first, trace_w as f32, trace_h as f32],
|
|
396
|
+
[surf_w as f32, surf_h as f32, 0.0, 0.0],
|
|
397
|
+
];
|
|
398
|
+
self.queue.write_buffer(&self.pt_atrous_params_bufs[i], 0, bytemuck::bytes_of(&p));
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
if self.pt_atrous_bgs[written_idx][0].is_none() {
|
|
402
|
+
let scratch = self.pt_atrous_scratch.as_ref().unwrap();
|
|
403
|
+
let scratch2 = self.pt_atrous_scratch2.as_ref().unwrap();
|
|
404
|
+
let accum_w = self.pt_accum_buffers[written_idx].as_ref().unwrap();
|
|
405
|
+
let moments_w = self.pt_moments_buffers[written_idx].as_ref().unwrap();
|
|
406
|
+
// Stage src → dst chain: accum→s1, then the scratches
|
|
407
|
+
// ping-pong; the final upsample reads the last-written
|
|
408
|
+
// scratch. cs_final never writes dst; it gets whichever
|
|
409
|
+
// scratch is not its src (RO+RW of one buffer in a
|
|
410
|
+
// single group fails validation).
|
|
411
|
+
let chain: [(&wgpu::Buffer, &wgpu::Buffer); 6] = [
|
|
412
|
+
(accum_w, scratch),
|
|
413
|
+
(scratch, scratch2),
|
|
414
|
+
(scratch2, scratch),
|
|
415
|
+
(scratch, scratch2),
|
|
416
|
+
(scratch2, scratch),
|
|
417
|
+
(scratch, scratch2),
|
|
418
|
+
];
|
|
419
|
+
for (i, (src, dst)) in chain.iter().enumerate() {
|
|
420
|
+
self.pt_atrous_bgs[written_idx][i] = Some(self.device.create_bind_group(&wgpu::BindGroupDescriptor {
|
|
421
|
+
label: Some("pt_atrous_bg"),
|
|
422
|
+
layout: self.pt_atrous_layout.as_ref().unwrap(),
|
|
423
|
+
entries: &[
|
|
424
|
+
wgpu::BindGroupEntry { binding: 0, resource: self.pt_atrous_params_bufs[i].as_entire_binding() },
|
|
425
|
+
wgpu::BindGroupEntry { binding: 1, resource: src.as_entire_binding() },
|
|
426
|
+
wgpu::BindGroupEntry { binding: 2, resource: dst.as_entire_binding() },
|
|
427
|
+
wgpu::BindGroupEntry { binding: 3, resource: wgpu::BindingResource::TextureView(&self.hdr_rt_view) },
|
|
428
|
+
wgpu::BindGroupEntry { binding: 4, resource: wgpu::BindingResource::TextureView(&self.depth_view) },
|
|
429
|
+
wgpu::BindGroupEntry { binding: 5, resource: wgpu::BindingResource::TextureView(&self.albedo_rt_view) },
|
|
430
|
+
wgpu::BindGroupEntry { binding: 6, resource: moments_w.as_entire_binding() },
|
|
431
|
+
],
|
|
432
|
+
}));
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
{
|
|
437
|
+
let ts = profiler.compute_pass_timestamp_writes("pt_atrous");
|
|
438
|
+
let mut pass = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor {
|
|
439
|
+
label: Some("pt_atrous"),
|
|
440
|
+
timestamp_writes: ts,
|
|
441
|
+
});
|
|
442
|
+
pass.set_pipeline(self.pt_atrous_mid_pipeline.as_ref().unwrap());
|
|
443
|
+
pass.set_bind_group(0, self.pt_atrous_bgs[written_idx][0].as_ref().unwrap(), &[]);
|
|
444
|
+
pass.dispatch_workgroups((trace_w + 7) / 8, (trace_h + 7) / 8, 1);
|
|
445
|
+
}
|
|
446
|
+
// History feedback: the pass split makes the copy legal
|
|
447
|
+
// (buffer copies cannot live inside a compute pass).
|
|
448
|
+
encoder.copy_buffer_to_buffer(
|
|
449
|
+
self.pt_atrous_scratch.as_ref().unwrap(),
|
|
450
|
+
0,
|
|
451
|
+
self.pt_accum_buffers[written_idx].as_ref().unwrap(),
|
|
452
|
+
0,
|
|
453
|
+
needed,
|
|
454
|
+
);
|
|
455
|
+
{
|
|
456
|
+
let ts = profiler.compute_pass_timestamp_writes("pt_atrous2");
|
|
457
|
+
let mut pass = encoder.begin_compute_pass(&wgpu::ComputePassDescriptor {
|
|
458
|
+
label: Some("pt_atrous2"),
|
|
459
|
+
timestamp_writes: ts,
|
|
460
|
+
});
|
|
461
|
+
pass.set_pipeline(self.pt_atrous_mid_pipeline.as_ref().unwrap());
|
|
462
|
+
for i in 1..5 {
|
|
463
|
+
pass.set_bind_group(0, self.pt_atrous_bgs[written_idx][i].as_ref().unwrap(), &[]);
|
|
464
|
+
pass.dispatch_workgroups((trace_w + 7) / 8, (trace_h + 7) / 8, 1);
|
|
465
|
+
}
|
|
466
|
+
pass.set_pipeline(self.pt_atrous_final_pipeline.as_ref().unwrap());
|
|
467
|
+
pass.set_bind_group(0, self.pt_atrous_bgs[written_idx][5].as_ref().unwrap(), &[]);
|
|
468
|
+
pass.dispatch_workgroups((surf_w + 7) / 8, (surf_h + 7) / 8, 1);
|
|
469
|
+
}
|
|
470
|
+
}
|
|
471
|
+
// Mirrors the kernel's write threshold: mode 1 leaves the raster
|
|
472
|
+
// frame on screen until 8 samples exist (u.size.w carried the
|
|
473
|
+
// pre-increment count), so SSGI/SSR must keep running for those
|
|
474
|
+
// frames — the gates downstream check pt_owns_frame().
|
|
475
|
+
self.pt_wrote_frame = self.pt_mode >= 2 || self.pt_accum_count >= 8;
|
|
476
|
+
self.pt_accum_count = self.pt_accum_count.saturating_add(1);
|
|
477
|
+
self.pt_frame_index = self.pt_frame_index.wrapping_add(1);
|
|
478
|
+
|
|
479
|
+
// ---- debug 16: numeric readback of traced intersections ----
|
|
480
|
+
// Copies a window of the accum buffer (center of frame) into a
|
|
481
|
+
// staging buffer each frame; the previous frame's copy is mapped
|
|
482
|
+
// (blocking) and dumped to pt_trace_dump.txt once.
|
|
483
|
+
if (self.pt_debug == 16.0
|
|
484
|
+
|| self.pt_debug == 17.0
|
|
485
|
+
|| self.pt_debug == 19.0
|
|
486
|
+
|| self.pt_debug == 22.0
|
|
487
|
+
|| self.pt_debug == 23.0)
|
|
488
|
+
&& self.pt_accum_count > 30
|
|
489
|
+
&& !self.pt_dump_written
|
|
490
|
+
{
|
|
491
|
+
// Offsets in TRACE-grid units — the accum buffers are
|
|
492
|
+
// half-res in realtime mode.
|
|
493
|
+
let dump_pixels: u64 = (trace_w as u64).min(4096);
|
|
494
|
+
let row = (trace_h / 2) as u64;
|
|
495
|
+
let offset = row * trace_w as u64 * 16;
|
|
496
|
+
if self.pt_readback_buffer.is_none() {
|
|
497
|
+
self.pt_readback_buffer = Some(self.device.create_buffer(&wgpu::BufferDescriptor {
|
|
498
|
+
label: Some("pt_readback"),
|
|
499
|
+
size: dump_pixels * 16,
|
|
500
|
+
usage: wgpu::BufferUsages::MAP_READ | wgpu::BufferUsages::COPY_DST,
|
|
501
|
+
mapped_at_creation: false,
|
|
502
|
+
}));
|
|
503
|
+
encoder.copy_buffer_to_buffer(
|
|
504
|
+
// written_idx == pt_accum_idx here (already flipped):
|
|
505
|
+
// the buffer this frame's dispatch wrote.
|
|
506
|
+
self.pt_accum_buffers[self.pt_accum_idx].as_ref().unwrap(),
|
|
507
|
+
offset,
|
|
508
|
+
self.pt_readback_buffer.as_ref().unwrap(),
|
|
509
|
+
0,
|
|
510
|
+
dump_pixels * 16,
|
|
511
|
+
);
|
|
512
|
+
} else {
|
|
513
|
+
// Previous frame's copy has been submitted; map it now.
|
|
514
|
+
let buf = self.pt_readback_buffer.as_ref().unwrap();
|
|
515
|
+
let slice = buf.slice(..);
|
|
516
|
+
slice.map_async(wgpu::MapMode::Read, |_| {});
|
|
517
|
+
let _ = self.device.poll(wgpu::PollType::Wait { submission_index: None, timeout: None });
|
|
518
|
+
let data = slice.get_mapped_range();
|
|
519
|
+
let vals: &[[f32; 4]] = bytemuck::cast_slice(&data);
|
|
520
|
+
let mut out = String::new();
|
|
521
|
+
out.push_str(&format!(
|
|
522
|
+
"middle row, {} pixels, mode {}\n",
|
|
523
|
+
vals.len(),
|
|
524
|
+
self.pt_debug
|
|
525
|
+
));
|
|
526
|
+
// Every 64th pixel across the full row. Field meaning:
|
|
527
|
+
// 16 = t / id / prim / kind; 17 = p0.xyz / raw depth.
|
|
528
|
+
for (i, v) in vals.iter().enumerate().step_by(64) {
|
|
529
|
+
out.push_str(&format!(
|
|
530
|
+
"col {i}: {:.4} {:.4} {:.4} {:.6}\n",
|
|
531
|
+
v[0], v[1], v[2], v[3]
|
|
532
|
+
));
|
|
533
|
+
}
|
|
534
|
+
// Also dump the CPU-side uniform inputs for comparison,
|
|
535
|
+
// plus the unprojection computed in BOTH multiply
|
|
536
|
+
// conventions. Whichever matches the GPU dump is what
|
|
537
|
+
// the shader effectively computed; the other (if sane)
|
|
538
|
+
// is the fix.
|
|
539
|
+
let ndc = [0.0f32, 0.0, 0.998647, 1.0];
|
|
540
|
+
let m = &self.current_inv_vp_matrix;
|
|
541
|
+
let mut h_col = [0.0f32; 4]; // h_i = sum_c m[c][i] * ndc[c]
|
|
542
|
+
let mut h_row = [0.0f32; 4]; // h_i = sum_c m[i][c] * ndc[c]
|
|
543
|
+
for i in 0..4 {
|
|
544
|
+
for c in 0..4 {
|
|
545
|
+
h_col[i] += m[c][i] * ndc[c];
|
|
546
|
+
h_row[i] += m[i][c] * ndc[c];
|
|
547
|
+
}
|
|
548
|
+
}
|
|
549
|
+
out.push_str(&format!(
|
|
550
|
+
"cpu cam_pos = {:?}\n\
|
|
551
|
+
unproject as columns: h={:?} p={:?}\n\
|
|
552
|
+
unproject transposed: h={:?} p={:?}\n",
|
|
553
|
+
self.current_camera_pos,
|
|
554
|
+
h_col,
|
|
555
|
+
[h_col[0] / h_col[3], h_col[1] / h_col[3], h_col[2] / h_col[3]],
|
|
556
|
+
h_row,
|
|
557
|
+
[h_row[0] / h_row[3], h_row[1] / h_row[3], h_row[2] / h_row[3]],
|
|
558
|
+
));
|
|
559
|
+
// Distinct instance-id count over the whole row.
|
|
560
|
+
let mut ids: Vec<i64> = vals
|
|
561
|
+
.iter()
|
|
562
|
+
.filter(|v| v[3] != 0.0)
|
|
563
|
+
.map(|v| v[1] as i64)
|
|
564
|
+
.collect();
|
|
565
|
+
ids.sort_unstable();
|
|
566
|
+
ids.dedup();
|
|
567
|
+
let misses = vals.iter().filter(|v| v[3] == 0.0).count();
|
|
568
|
+
out.push_str(&format!("distinct hit ids: {:?}\n", ids));
|
|
569
|
+
out.push_str(&format!("misses: {}\n", misses));
|
|
570
|
+
drop(data);
|
|
571
|
+
buf.unmap();
|
|
572
|
+
let _ = std::fs::write("pt_trace_dump.txt", out);
|
|
573
|
+
self.pt_dump_written = true;
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
}
|
|
577
|
+
}
|