@bornengine/engine 0.4.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +231 -0
- package/native/android/Cargo.lock +1848 -0
- package/native/android/Cargo.toml +24 -0
- package/native/android/src/lib.rs +702 -0
- package/native/ios/Cargo.lock +1690 -0
- package/native/ios/Cargo.toml +32 -0
- package/native/ios/src/lib.rs +1267 -0
- package/native/linux/Cargo.lock +3279 -0
- package/native/linux/Cargo.toml +29 -0
- package/native/linux/src/lib.rs +1331 -0
- package/native/macos/Cargo.lock +3310 -0
- package/native/macos/Cargo.toml +46 -0
- package/native/macos/src/lib.rs +1302 -0
- package/native/shared/Cargo.lock +1899 -0
- package/native/shared/Cargo.toml +62 -0
- package/native/shared/assets/default_font.ttf +0 -0
- package/native/shared/build.rs +270 -0
- package/native/shared/shaders/common/clouds.wgsl +122 -0
- package/native/shared/shaders/common/fog.wgsl +16 -0
- package/native/shared/shaders/common/foliage_wind.wgsl +98 -0
- package/native/shared/shaders/common/imposter.wgsl +112 -0
- package/native/shared/shaders/common/pbr.wgsl +186 -0
- package/native/shared/shaders/common/shadows.wgsl +186 -0
- package/native/shared/shaders/common/sky.wgsl +8 -0
- package/native/shared/shaders/common/tonemap.wgsl +25 -0
- package/native/shared/shaders/impulse_field.wgsl +57 -0
- package/native/shared/shaders/material_abi.wgsl +383 -0
- package/native/shared/shaders/materials/test_minimal.wgsl +42 -0
- package/native/shared/src/anim_mixer.rs +61 -0
- package/native/shared/src/attach.rs +263 -0
- package/native/shared/src/audio/decode.rs +123 -0
- package/native/shared/src/audio/mod.rs +863 -0
- package/native/shared/src/audio/render.rs +892 -0
- package/native/shared/src/audio/spsc.rs +156 -0
- package/native/shared/src/audio/stream.rs +226 -0
- package/native/shared/src/custom_shaders.rs +104 -0
- package/native/shared/src/decals.rs +245 -0
- package/native/shared/src/drs.rs +211 -0
- package/native/shared/src/engine.rs +261 -0
- package/native/shared/src/ffi.rs +116 -0
- package/native/shared/src/ffi_core/assets.rs +388 -0
- package/native/shared/src/ffi_core/audio_ffi.rs +184 -0
- package/native/shared/src/ffi_core/draw.rs +334 -0
- package/native/shared/src/ffi_core/game_loop.rs +577 -0
- package/native/shared/src/ffi_core/input.rs +234 -0
- package/native/shared/src/ffi_core/mod.rs +127 -0
- package/native/shared/src/ffi_core/models.rs +1154 -0
- package/native/shared/src/ffi_core/ragdoll_ffi.rs +261 -0
- package/native/shared/src/ffi_core/scene.rs +626 -0
- package/native/shared/src/ffi_core/vfx.rs +212 -0
- package/native/shared/src/ffi_core/visual.rs +691 -0
- package/native/shared/src/frame_callbacks.rs +122 -0
- package/native/shared/src/geometry.rs +236 -0
- package/native/shared/src/handles.rs +182 -0
- package/native/shared/src/input.rs +448 -0
- package/native/shared/src/jolt_sys.rs +822 -0
- package/native/shared/src/lib.rs +55 -0
- package/native/shared/src/models.rs +1093 -0
- package/native/shared/src/models_gltf.rs +1280 -0
- package/native/shared/src/particles.rs +391 -0
- package/native/shared/src/physics_jolt.rs +1908 -0
- package/native/shared/src/picking.rs +298 -0
- package/native/shared/src/postfx.rs +345 -0
- package/native/shared/src/profiler.rs +492 -0
- package/native/shared/src/ragdoll.rs +474 -0
- package/native/shared/src/renderer/atmosphere_lut.rs +573 -0
- package/native/shared/src/renderer/brdf_lut.rs +154 -0
- package/native/shared/src/renderer/draw2d.rs +143 -0
- package/native/shared/src/renderer/formats.rs +822 -0
- package/native/shared/src/renderer/froxel.rs +421 -0
- package/native/shared/src/renderer/gi_bake.rs +653 -0
- package/native/shared/src/renderer/graph.rs +462 -0
- package/native/shared/src/renderer/hiz.rs +269 -0
- package/native/shared/src/renderer/hot_reload.rs +390 -0
- package/native/shared/src/renderer/impulse_field.rs +456 -0
- package/native/shared/src/renderer/lighting.rs +154 -0
- package/native/shared/src/renderer/material_instancing.rs +171 -0
- package/native/shared/src/renderer/material_pipeline.rs +700 -0
- package/native/shared/src/renderer/material_system.rs +1996 -0
- package/native/shared/src/renderer/material_system_tests.rs +601 -0
- package/native/shared/src/renderer/material_system_wasm.rs +41 -0
- package/native/shared/src/renderer/mod.rs +12556 -0
- package/native/shared/src/renderer/model_draw.rs +641 -0
- package/native/shared/src/renderer/occlusion.rs +429 -0
- package/native/shared/src/renderer/planar_pass.rs +593 -0
- package/native/shared/src/renderer/planar_reflection.rs +499 -0
- package/native/shared/src/renderer/post_pass.rs +249 -0
- package/native/shared/src/renderer/postfx_chain.rs +728 -0
- package/native/shared/src/renderer/pt_pass.rs +577 -0
- package/native/shared/src/renderer/scene_pass.rs +607 -0
- package/native/shared/src/renderer/shader_include.rs +205 -0
- package/native/shared/src/renderer/shader_library.rs +135 -0
- package/native/shared/src/renderer/shaders/ao.rs +570 -0
- package/native/shared/src/renderer/shaders/core.rs +1243 -0
- package/native/shared/src/renderer/shaders/env.rs +907 -0
- package/native/shared/src/renderer/shaders/gi.rs +810 -0
- package/native/shared/src/renderer/shaders/mod.rs +19 -0
- package/native/shared/src/renderer/shaders/post.rs +1558 -0
- package/native/shared/src/renderer/shaders/pt.rs +1859 -0
- package/native/shared/src/renderer/shaders/ssgi.rs +1586 -0
- package/native/shared/src/renderer/shadow_pass.rs +731 -0
- package/native/shared/src/renderer/ssgi_pass.rs +392 -0
- package/native/shared/src/renderer/ssr_pass.rs +188 -0
- package/native/shared/src/renderer/texture_store.rs +473 -0
- package/native/shared/src/renderer/transient.rs +591 -0
- package/native/shared/src/renderer/types.rs +941 -0
- package/native/shared/src/renderer/util.rs +152 -0
- package/native/shared/src/scene.rs +1362 -0
- package/native/shared/src/sdf_cache.rs +274 -0
- package/native/shared/src/shadows.rs +1036 -0
- package/native/shared/src/staging.rs +102 -0
- package/native/shared/src/string_header.rs +266 -0
- package/native/shared/src/text_renderer.rs +502 -0
- package/native/shared/src/textures.rs +197 -0
- package/native/tvos/Cargo.lock +1693 -0
- package/native/tvos/Cargo.toml +36 -0
- package/native/tvos/metal-patched/Cargo.toml +178 -0
- package/native/tvos/metal-patched/LICENSE-APACHE +201 -0
- package/native/tvos/metal-patched/LICENSE-MIT +25 -0
- package/native/tvos/metal-patched/src/acceleration_structure.rs +667 -0
- package/native/tvos/metal-patched/src/acceleration_structure_pass.rs +108 -0
- package/native/tvos/metal-patched/src/argument.rs +366 -0
- package/native/tvos/metal-patched/src/blitpass.rs +102 -0
- package/native/tvos/metal-patched/src/buffer.rs +71 -0
- package/native/tvos/metal-patched/src/capturedescriptor.rs +76 -0
- package/native/tvos/metal-patched/src/capturemanager.rs +113 -0
- package/native/tvos/metal-patched/src/commandbuffer.rs +192 -0
- package/native/tvos/metal-patched/src/commandqueue.rs +44 -0
- package/native/tvos/metal-patched/src/computepass.rs +107 -0
- package/native/tvos/metal-patched/src/constants.rs +152 -0
- package/native/tvos/metal-patched/src/counters.rs +119 -0
- package/native/tvos/metal-patched/src/depthstencil.rs +190 -0
- package/native/tvos/metal-patched/src/device.rs +2134 -0
- package/native/tvos/metal-patched/src/drawable.rs +39 -0
- package/native/tvos/metal-patched/src/encoder.rs +2041 -0
- package/native/tvos/metal-patched/src/heap.rs +281 -0
- package/native/tvos/metal-patched/src/indirect_encoder.rs +344 -0
- package/native/tvos/metal-patched/src/lib.rs +657 -0
- package/native/tvos/metal-patched/src/library.rs +902 -0
- package/native/tvos/metal-patched/src/mps.rs +575 -0
- package/native/tvos/metal-patched/src/pipeline/compute.rs +475 -0
- package/native/tvos/metal-patched/src/pipeline/mod.rs +71 -0
- package/native/tvos/metal-patched/src/pipeline/render.rs +762 -0
- package/native/tvos/metal-patched/src/renderpass.rs +443 -0
- package/native/tvos/metal-patched/src/resource.rs +182 -0
- package/native/tvos/metal-patched/src/sampler.rs +165 -0
- package/native/tvos/metal-patched/src/sync.rs +178 -0
- package/native/tvos/metal-patched/src/texture.rs +352 -0
- package/native/tvos/metal-patched/src/types.rs +90 -0
- package/native/tvos/metal-patched/src/vertexdescriptor.rs +250 -0
- package/native/tvos/src/audio_backend.rs +197 -0
- package/native/tvos/src/lib.rs +1891 -0
- package/native/visionos/Cargo.lock +1693 -0
- package/native/visionos/Cargo.toml +40 -0
- package/native/visionos/src/audio_backend.rs +197 -0
- package/native/visionos/src/lib.rs +1887 -0
- package/native/watchos/Cargo.lock +16 -0
- package/native/watchos/Cargo.toml +19 -0
- package/native/watchos/shaders/bloom_postfx.metal +99 -0
- package/native/watchos/src/BloomWatchApp.swift +1267 -0
- package/native/watchos/src/BloomWatchAudio.swift +179 -0
- package/native/watchos/src/audio.rs +55 -0
- package/native/watchos/src/draw_list.rs +229 -0
- package/native/watchos/src/ffi_stubs.rs +915 -0
- package/native/watchos/src/ffi_stubs_manual.rs +35 -0
- package/native/watchos/src/lib.rs +1124 -0
- package/native/watchos/src/models.rs +746 -0
- package/native/watchos/src/postfx.rs +95 -0
- package/native/watchos/src/scene.rs +534 -0
- package/native/watchos/src/textures.rs +184 -0
- package/native/web/Cargo.lock +1657 -0
- package/native/web/Cargo.toml +43 -0
- package/native/web/bloom_glue.js +695 -0
- package/native/web/build.sh +131 -0
- package/native/web/index.html +35 -0
- package/native/web/jolt_bridge.js +1519 -0
- package/native/web/src/input_ffi.rs +286 -0
- package/native/web/src/lib.rs +1796 -0
- package/native/web/src/material_ffi.rs +710 -0
- package/native/web/src/parity_ffi.rs +343 -0
- package/native/web/src/physics_ffi.rs +643 -0
- package/native/web/src/ragdoll_ffi.rs +250 -0
- package/native/web/src/render_settings.rs +98 -0
- package/native/windows/Cargo.lock +1815 -0
- package/native/windows/Cargo.toml +68 -0
- package/native/windows/src/lib.rs +1486 -0
- package/package.json +4279 -0
- package/src/audio/index.ts +315 -0
- package/src/core/colors.ts +63 -0
- package/src/core/index.ts +1206 -0
- package/src/core/keys.ts +63 -0
- package/src/core/types.ts +104 -0
- package/src/index.ts +171 -0
- package/src/math/index.ts +516 -0
- package/src/mobile/index.ts +294 -0
- package/src/models/index.ts +1258 -0
- package/src/physics/index.ts +1134 -0
- package/src/scene/index.ts +698 -0
- package/src/shapes/index.ts +120 -0
- package/src/text/index.ts +48 -0
- package/src/textures/index.ts +187 -0
- package/src/vfx/index.ts +191 -0
- package/src/world/index.ts +24 -0
- package/src/world/loader.ts +423 -0
- package/src/world/prefab.ts +217 -0
- package/src/world/render.ts +172 -0
- package/src/world/saver.ts +108 -0
- package/src/world/serialize.ts +301 -0
- package/src/world/terrain.ts +355 -0
- package/src/world/types.ts +160 -0
- package/src/world/validate.ts +319 -0
- package/src/world/version.ts +114 -0
|
@@ -0,0 +1,1859 @@
|
|
|
1
|
+
//! Path-tracing megakernel (docs/pt/pt-roadmap.md, ticket PT-1).
|
|
2
|
+
//!
|
|
3
|
+
//! One compute kernel, one ray budget per pixel per frame. Primary hits come
|
|
4
|
+
//! from the G-buffer (depth + albedo + material MRTs — free, and sharper than
|
|
5
|
+
//! traced primaries); bounce and shadow rays go through the same TLAS the
|
|
6
|
+
//! Lumen HW probe trace uses. Hit shading at bounces reads the mesh-card
|
|
7
|
+
//! ALBEDO atlas (not the pre-lit radiance atlas: a path tracer computes its
|
|
8
|
+
//! own lighting at every vertex of the path — sampling pre-lit cards would
|
|
9
|
+
//! bake Lumen's direct light into ours twice).
|
|
10
|
+
//!
|
|
11
|
+
//! Radiometric convention: light intensities are treated as π-premultiplied,
|
|
12
|
+
//! i.e. diffuse contribution is `albedo * L * NdotL` with no 1/π — matching
|
|
13
|
+
//! the raster shader (core.rs point-light loop has no 1/π either), so
|
|
14
|
+
//! toggling PT on/off does not jump scene brightness. bloom-reference
|
|
15
|
+
//! comparisons account for this in scene config (see the PT-1 ticket).
|
|
16
|
+
//!
|
|
17
|
+
//! Sky pixels are never written: the raster sky/cloud passes already drew
|
|
18
|
+
//! them, and PT replacing a procedural cloud deck with an analytic gradient
|
|
19
|
+
//! would be a downgrade. PT owns geometry pixels only. The translucent pass
|
|
20
|
+
//! runs AFTER this kernel, so water and glass composite over path-traced
|
|
21
|
+
//! opaques exactly as they do over raster ones.
|
|
22
|
+
//!
|
|
23
|
+
//! Debug modes (uniform cfg.w, set via BLOOM_PT_DEBUG):
|
|
24
|
+
//! 1 = raw depth visualised 2 = reconstructed world normals
|
|
25
|
+
//! 3 = G-buffer albedo 4 = sun shadow-ray visibility
|
|
26
|
+
//! 5 = solid magenta (pipeline probe — proves dispatch + write path)
|
|
27
|
+
//! 6 = traced-primary interpolated normal (compare against 2;
|
|
28
|
+
//! magenta = TLAS miss where G-buffer had geometry, orange = hit
|
|
29
|
+
//! instance without a geometry window)
|
|
30
|
+
//! 7 = traced-primary textured hit albedo (compare against 3;
|
|
31
|
+
//! yellow = adapter lacks texture-array features)
|
|
32
|
+
//! 8-15 = binary/quantized probes from the DX12 bring-up (hit-window
|
|
33
|
+
//! flag, normal axes, primitive/instance banding, two-query
|
|
34
|
+
//! aliasing, t-vs-G-buffer sanity, t contours). 13 is the
|
|
35
|
+
//! keeper: green = traced t agrees with the G-buffer, red =
|
|
36
|
+
//! mismatch, blue = miss.
|
|
37
|
+
//! 16/17 = NUMERIC dumps via the accum buffer + CPU readback
|
|
38
|
+
//! (pt_trace_dump.txt): 16 = t/instance/prim/kind, 17 = p0 +
|
|
39
|
+
//! raw depth. These found the transposed inv_vp: when every
|
|
40
|
+
//! probe looks "constant", dump numbers before theorizing.
|
|
41
|
+
|
|
42
|
+
pub(in crate::renderer) const PT_KERNEL_WGSL: &str = r#"
|
|
43
|
+
struct PtLight {
|
|
44
|
+
pos_range: vec4<f32>, // xyz world position, w = range
|
|
45
|
+
color_int: vec4<f32>, // rgb color, w = intensity
|
|
46
|
+
};
|
|
47
|
+
|
|
48
|
+
struct PtParams {
|
|
49
|
+
inv_vp: mat4x4<f32>,
|
|
50
|
+
// PT-3: previous frame's UNJITTERED view-projection — reprojects
|
|
51
|
+
// this frame's world positions into last frame's screen for
|
|
52
|
+
// temporal history fetch in realtime mode.
|
|
53
|
+
prev_vp: mat4x4<f32>,
|
|
54
|
+
cam_pos: vec4<f32>, // xyz camera world pos
|
|
55
|
+
sun_dir: vec4<f32>, // xyz unit vector toward the sun
|
|
56
|
+
sun_color: vec4<f32>, // rgb premultiplied by intensity
|
|
57
|
+
sky_color: vec4<f32>, // rgb ambient-derived sky tint
|
|
58
|
+
size: vec4<u32>, // x/y = TRACE grid dims, z=frame_index, w=accum_count
|
|
59
|
+
cfg: vec4<f32>, // x=mode(1|2), y=max_bounces, z=point_light_count, w=debug
|
|
60
|
+
// PT-3 half-res: x/y = full G-buffer dims. z = 1 -> hybrid sun
|
|
61
|
+
// (sample the raster shadow cascades instead of tracing the sun;
|
|
62
|
+
// crisp noise-free direct shadows, rays spent on indirect only).
|
|
63
|
+
// w = 1 -> ReSTIR DI (PT-4, experimental).
|
|
64
|
+
ext: vec4<u32>,
|
|
65
|
+
// Raster shadow cascade view-projections. Uploaded RAW, like
|
|
66
|
+
// prev_vp: mat4_multiply products are already in WGSL M*v layout;
|
|
67
|
+
// only mat4_invert outputs (inv_vp) upload transposed.
|
|
68
|
+
shadow_vps: array<mat4x4<f32>, 3>,
|
|
69
|
+
lights: array<PtLight, 16>,
|
|
70
|
+
};
|
|
71
|
+
|
|
72
|
+
// Layout mirror of the Lumen instance data (ssgi.rs) — same buffer.
|
|
73
|
+
struct InstanceGiData {
|
|
74
|
+
albedo: vec3<f32>,
|
|
75
|
+
emissive_luma: f32,
|
|
76
|
+
normal_ws: vec3<f32>,
|
|
77
|
+
_pad0: f32,
|
|
78
|
+
card_slot: vec4<f32>,
|
|
79
|
+
card_aabb_min: vec4<f32>,
|
|
80
|
+
card_aabb_max: vec4<f32>,
|
|
81
|
+
world_aabb_min: vec4<f32>,
|
|
82
|
+
world_aabb_max: vec4<f32>,
|
|
83
|
+
// PT-2: x = vertex_base, y = index_base, z = index_count (0 = no
|
|
84
|
+
// geometry window -> PT-1 fallback), w = albedo texture index.
|
|
85
|
+
geo: vec4<u32>,
|
|
86
|
+
// PT-2: x = roughness, y = metalness.
|
|
87
|
+
mat_params: vec4<f32>,
|
|
88
|
+
};
|
|
89
|
+
|
|
90
|
+
@group(0) @binding(0) var<uniform> u: PtParams;
|
|
91
|
+
@group(0) @binding(1) var accel: acceleration_structure;
|
|
92
|
+
@group(0) @binding(2) var<storage, read> instance_data: array<InstanceGiData>;
|
|
93
|
+
@group(0) @binding(3) var depth_tex: texture_depth_2d;
|
|
94
|
+
@group(0) @binding(4) var albedo_tex: texture_2d<f32>;
|
|
95
|
+
@group(0) @binding(5) var material_tex: texture_2d<f32>;
|
|
96
|
+
@group(0) @binding(6) var card_albedo_atlas: texture_2d<f32>;
|
|
97
|
+
@group(0) @binding(7) var card_samp: sampler;
|
|
98
|
+
// PT-3: ping-pong accumulation. Binding 8 = previous frame's buffer
|
|
99
|
+
// (read), binding 13 = this frame's output. Reprojection reads OTHER
|
|
100
|
+
// pixels from prev, which a single read_write buffer cannot do safely.
|
|
101
|
+
//
|
|
102
|
+
// Layout (SVGF): accum = (irradiance rgb, luminance variance);
|
|
103
|
+
// moments = (mu1, mu2, history length, raw depth). Progressive mode
|
|
104
|
+
// keeps its original (radiance sum, sample count) layout in accum and
|
|
105
|
+
// leaves the moments buffers untouched.
|
|
106
|
+
@group(0) @binding(8) var<storage, read_write> accum: array<vec4<f32>>;
|
|
107
|
+
@group(0) @binding(9) var out_hdr: texture_storage_2d<rgba16float, write>;
|
|
108
|
+
@group(0) @binding(13) var<storage, read_write> accum_out: array<vec4<f32>>;
|
|
109
|
+
@group(0) @binding(18) var<storage, read_write> moments: array<vec4<f32>>;
|
|
110
|
+
@group(0) @binding(19) var<storage, read_write> moments_out: array<vec4<f32>>;
|
|
111
|
+
// PT-4 (EXPERIMENTAL, ext.w == 1) — ReSTIR DI reservoirs, ping-pong
|
|
112
|
+
// with the accum pair: (light index, W, M, target pdf) per trace texel.
|
|
113
|
+
@group(0) @binding(20) var<storage, read_write> resv: array<vec4<f32>>;
|
|
114
|
+
@group(0) @binding(21) var<storage, read_write> resv_out: array<vec4<f32>>;
|
|
115
|
+
// PT-7 — the raster velocity MRT (uv-space delta, current − previous,
|
|
116
|
+
// no Y flip at write; see core.rs). Non-zero where a surface MOVED —
|
|
117
|
+
// reprojection follows it instead of the camera-only prev_vp math,
|
|
118
|
+
// exactly like TAA, so moving skinned characters keep their history.
|
|
119
|
+
@group(0) @binding(22) var velocity_tex: texture_2d<f32>;
|
|
120
|
+
|
|
121
|
+
// Reprojection of the current surface into the previous frame's trace
|
|
122
|
+
// grid — computed once per texel (module privates because both the
|
|
123
|
+
// ReSTIR temporal reuse and the SVGF colour accumulation consume it).
|
|
124
|
+
var<private> rp_valid: bool;
|
|
125
|
+
var<private> rp_base: vec2<i32>;
|
|
126
|
+
var<private> rp_fr: vec2<f32>;
|
|
127
|
+
var<private> rp_zl_here: f32;
|
|
128
|
+
var<private> rp_nearest: u32;
|
|
129
|
+
|
|
130
|
+
fn compute_reproj(p0: vec3<f32>, px_full: vec2<i32>, depth_cur: f32) {
|
|
131
|
+
rp_valid = false;
|
|
132
|
+
if (u.size.w == 0u) { return; }
|
|
133
|
+
// PT-7 — object motion first: the velocity buffer knows how THIS
|
|
134
|
+
// pixel's surface moved (including skeletal motion, which no
|
|
135
|
+
// camera matrix can express). TAA's convention:
|
|
136
|
+
// prev_uv = (uv.x - vel.x, uv.y + vel.y). Camera-only pixels
|
|
137
|
+
// write ~zero velocity and fall through to the prev_vp math.
|
|
138
|
+
var uv_prev: vec2<f32>;
|
|
139
|
+
var zl_here: f32;
|
|
140
|
+
let vel = textureLoad(velocity_tex, px_full, 0).rg;
|
|
141
|
+
if (abs(vel.x) + abs(vel.y) > 1e-5) {
|
|
142
|
+
let uv_cur = (vec2<f32>(px_full) + 0.5)
|
|
143
|
+
/ vec2<f32>(f32(u.ext.x), f32(u.ext.y));
|
|
144
|
+
uv_prev = vec2<f32>(uv_cur.x - vel.x, uv_cur.y + vel.y);
|
|
145
|
+
// Depth along a moving surface changes slowly frame-to-frame;
|
|
146
|
+
// the tap tolerance absorbs it, and a fast approach degrades
|
|
147
|
+
// to a disocclusion reset — the safe direction.
|
|
148
|
+
zl_here = lin_depth(depth_cur);
|
|
149
|
+
} else {
|
|
150
|
+
let clip_prev = u.prev_vp * vec4<f32>(p0, 1.0);
|
|
151
|
+
if (clip_prev.w <= 1e-4) { return; }
|
|
152
|
+
let ndc_prev = clip_prev.xyz / clip_prev.w;
|
|
153
|
+
uv_prev = vec2<f32>(ndc_prev.x * 0.5 + 0.5, 0.5 - ndc_prev.y * 0.5);
|
|
154
|
+
zl_here = lin_depth(ndc_prev.z);
|
|
155
|
+
}
|
|
156
|
+
if (uv_prev.x < 0.0 || uv_prev.x >= 1.0 || uv_prev.y < 0.0 || uv_prev.y >= 1.0) {
|
|
157
|
+
return;
|
|
158
|
+
}
|
|
159
|
+
let pos = uv_prev * vec2<f32>(f32(u.size.x), f32(u.size.y)) - 0.5;
|
|
160
|
+
rp_base = vec2<i32>(floor(pos));
|
|
161
|
+
rp_fr = pos - floor(pos);
|
|
162
|
+
rp_zl_here = zl_here;
|
|
163
|
+
let np = vec2<u32>(
|
|
164
|
+
min(u32(max(rp_base.x + i32(round(rp_fr.x)), 0)), u.size.x - 1u),
|
|
165
|
+
min(u32(max(rp_base.y + i32(round(rp_fr.y)), 0)), u.size.y - 1u),
|
|
166
|
+
);
|
|
167
|
+
rp_nearest = np.y * u.size.x + np.x;
|
|
168
|
+
rp_valid = true;
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
// Approximate linear view distance from the raw depth-buffer value
|
|
172
|
+
// (GL-convention matrix, near 0.01: z_view ~= 2n / (1 - d)). Only used
|
|
173
|
+
// for RELATIVE history-validation comparisons.
|
|
174
|
+
fn lin_depth(d: f32) -> f32 {
|
|
175
|
+
return 0.02 / max(1.0 - d, 1e-6);
|
|
176
|
+
}
|
|
177
|
+
// PT-2: geometry megabuffers. geo_v holds raw Vertex3D words (stride 24
|
|
178
|
+
// f32: position +0, normal +3, color +6, uv +10, ...); geo_i holds the
|
|
179
|
+
// concatenated index streams. Windows are per-instance via inst.geo.
|
|
180
|
+
// (Binding 12, the texture array + PT_HAS_TEXTURES + pt_tex_sample, is
|
|
181
|
+
// appended by the Rust side per adapter support.)
|
|
182
|
+
@group(0) @binding(10) var<storage, read> geo_v: array<f32>;
|
|
183
|
+
@group(0) @binding(11) var<storage, read> geo_i: array<u32>;
|
|
184
|
+
// Hybrid sun (ext.z == 1): the raster shadow cascades.
|
|
185
|
+
@group(0) @binding(14) var shadow_atlas_0: texture_depth_2d;
|
|
186
|
+
@group(0) @binding(15) var shadow_atlas_1: texture_depth_2d;
|
|
187
|
+
@group(0) @binding(16) var shadow_atlas_2: texture_depth_2d;
|
|
188
|
+
@group(0) @binding(17) var shadow_samp: sampler_comparison;
|
|
189
|
+
|
|
190
|
+
// Sun visibility from the shadow cascades (near -> far fallthrough by
|
|
191
|
+
// coverage, same scheme as the WSRC bake). Deterministic and smooth —
|
|
192
|
+
// the whole reason RT mode's direct light doesn't shimmer or dither.
|
|
193
|
+
fn sun_vis_cascade(pos_ws: vec3<f32>) -> f32 {
|
|
194
|
+
for (var c = 0; c < 3; c = c + 1) {
|
|
195
|
+
var clip: vec4<f32>;
|
|
196
|
+
if (c == 0) { clip = u.shadow_vps[0] * vec4<f32>(pos_ws, 1.0); }
|
|
197
|
+
else if (c == 1) { clip = u.shadow_vps[1] * vec4<f32>(pos_ws, 1.0); }
|
|
198
|
+
else { clip = u.shadow_vps[2] * vec4<f32>(pos_ws, 1.0); }
|
|
199
|
+
if (abs(clip.w) < 1e-6) { continue; }
|
|
200
|
+
let ndc = clip.xyz / clip.w;
|
|
201
|
+
if (ndc.x < -0.99 || ndc.x > 0.99 || ndc.y < -0.99 || ndc.y > 0.99 || ndc.z < 0.0 || ndc.z > 1.0) {
|
|
202
|
+
continue;
|
|
203
|
+
}
|
|
204
|
+
let uv = vec2<f32>(ndc.x * 0.5 + 0.5, 0.5 - ndc.y * 0.5);
|
|
205
|
+
let ref_depth = ndc.z - 0.002;
|
|
206
|
+
// Manual load-and-compare with a 2x2 average instead of the
|
|
207
|
+
// comparison sampler: SampleCmp from a COMPUTE stage proved
|
|
208
|
+
// unreliable on this DXC path (constant 0, independent of the
|
|
209
|
+
// matrices — same failure shape as the ray-query saga), and the
|
|
210
|
+
// only other compute-stage user (WSRC bake) was never validated
|
|
211
|
+
// on DX12. textureLoad is proven (the PT depth reads use it).
|
|
212
|
+
var dims: vec2<u32>;
|
|
213
|
+
if (c == 0) { dims = textureDimensions(shadow_atlas_0); }
|
|
214
|
+
else if (c == 1) { dims = textureDimensions(shadow_atlas_1); }
|
|
215
|
+
else { dims = textureDimensions(shadow_atlas_2); }
|
|
216
|
+
let fdims = vec2<f32>(dims);
|
|
217
|
+
var vis = 0.0;
|
|
218
|
+
for (var ty = 0; ty <= 1; ty = ty + 1) {
|
|
219
|
+
for (var tx = 0; tx <= 1; tx = tx + 1) {
|
|
220
|
+
let tc = clamp(
|
|
221
|
+
vec2<i32>(uv * fdims - vec2<f32>(0.5)) + vec2<i32>(tx, ty),
|
|
222
|
+
vec2<i32>(0),
|
|
223
|
+
vec2<i32>(i32(dims.x) - 1, i32(dims.y) - 1),
|
|
224
|
+
);
|
|
225
|
+
var stored: f32;
|
|
226
|
+
if (c == 0) { stored = textureLoad(shadow_atlas_0, tc, 0); }
|
|
227
|
+
else if (c == 1) { stored = textureLoad(shadow_atlas_1, tc, 0); }
|
|
228
|
+
else { stored = textureLoad(shadow_atlas_2, tc, 0); }
|
|
229
|
+
if (ref_depth <= stored) { vis += 0.25; }
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
return vis;
|
|
233
|
+
}
|
|
234
|
+
return 1.0;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
const PT_VSTRIDE: u32 = 24u;
|
|
238
|
+
|
|
239
|
+
struct HitAttrs {
|
|
240
|
+
normal_os: vec3<f32>,
|
|
241
|
+
uv: vec2<f32>,
|
|
242
|
+
};
|
|
243
|
+
|
|
244
|
+
fn vert_normal_os(slot: u32) -> vec3<f32> {
|
|
245
|
+
let o = slot * PT_VSTRIDE + 3u;
|
|
246
|
+
return vec3<f32>(geo_v[o], geo_v[o + 1u], geo_v[o + 2u]);
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
fn vert_uv(slot: u32) -> vec2<f32> {
|
|
250
|
+
let o = slot * PT_VSTRIDE + 10u;
|
|
251
|
+
return vec2<f32>(geo_v[o], geo_v[o + 1u]);
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
// Interpolate the hit triangle's vertex normal + UV. DXR/Vulkan
|
|
255
|
+
// barycentric convention: (u, v) weight vertices 1 and 2, w = 1-u-v
|
|
256
|
+
// weights vertex 0.
|
|
257
|
+
fn fetch_hit_attrs(geo: vec4<u32>, prim: u32, bary: vec2<f32>) -> HitAttrs {
|
|
258
|
+
let base = geo.y + prim * 3u;
|
|
259
|
+
let s0 = geo.x + geo_i[base];
|
|
260
|
+
let s1 = geo.x + geo_i[base + 1u];
|
|
261
|
+
let s2 = geo.x + geo_i[base + 2u];
|
|
262
|
+
let w = 1.0 - bary.x - bary.y;
|
|
263
|
+
var a: HitAttrs;
|
|
264
|
+
a.normal_os = w * vert_normal_os(s0) + bary.x * vert_normal_os(s1) + bary.y * vert_normal_os(s2);
|
|
265
|
+
a.uv = w * vert_uv(s0) + bary.x * vert_uv(s1) + bary.y * vert_uv(s2);
|
|
266
|
+
return a;
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
// Object-space normal -> world space: with M = object_to_world the
|
|
270
|
+
// correct transform is (M^-1)^T, and the ray query hands us M^-1 as
|
|
271
|
+
// world_to_object. `v * mat3` multiplies by the transpose in WGSL.
|
|
272
|
+
fn normal_to_world(n_os: vec3<f32>, w2o: mat4x3<f32>) -> vec3<f32> {
|
|
273
|
+
let lin = mat3x3<f32>(w2o[0], w2o[1], w2o[2]);
|
|
274
|
+
let n = n_os * lin;
|
|
275
|
+
let len = length(n);
|
|
276
|
+
if (len < 1e-8) { return vec3<f32>(0.0, 1.0, 0.0); }
|
|
277
|
+
return n / len;
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
// ---- RNG: PCG, one stream per (pixel, frame) --------------------------------
|
|
281
|
+
|
|
282
|
+
var<private> rng_state: u32;
|
|
283
|
+
|
|
284
|
+
fn rng_seed(px: vec2<u32>, frame: u32) {
|
|
285
|
+
var h = px.x * 374761393u + px.y * 668265263u + frame * 2654435761u;
|
|
286
|
+
h = (h ^ (h >> 13u)) * 1274126177u;
|
|
287
|
+
rng_state = h ^ (h >> 16u);
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
fn rand_f() -> f32 {
|
|
291
|
+
// PCG-XSH-RR step.
|
|
292
|
+
let old = rng_state;
|
|
293
|
+
rng_state = old * 747796405u + 2891336453u;
|
|
294
|
+
let word = ((old >> ((old >> 28u) + 4u)) ^ old) * 277803737u;
|
|
295
|
+
let out = (word >> 22u) ^ word;
|
|
296
|
+
return f32(out) * 2.3283064e-10; // / 2^32
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
fn rand_2f() -> vec2<f32> { return vec2<f32>(rand_f(), rand_f()); }
|
|
300
|
+
|
|
301
|
+
// Interleaved gradient noise, scrolled per frame by the golden-ratio
|
|
302
|
+
// offset. Spatially STRUCTURED (neighbors get well-distributed values)
|
|
303
|
+
// so a 5x5 filter averages it nearly flat — white PCG noise leaves
|
|
304
|
+
// mid-frequency blotch at 1-2 spp that shimmers. Used for the primary
|
|
305
|
+
// sun test in realtime mode.
|
|
306
|
+
fn ign_at(px: vec2<i32>, frame: u32) -> f32 {
|
|
307
|
+
let p = vec2<f32>(px) + f32(frame % 64u) * 5.588238;
|
|
308
|
+
return fract(52.9829189 * fract(0.06711056 * p.x + 0.00583715 * p.y));
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
// ---- Geometry reconstruction -------------------------------------------------
|
|
312
|
+
|
|
313
|
+
// px here is always a FULL-resolution G-buffer pixel (u.ext dims); the
|
|
314
|
+
// trace grid may be half of that in realtime mode.
|
|
315
|
+
fn world_at(px: vec2<i32>, depth: f32) -> vec3<f32> {
|
|
316
|
+
let dims = vec2<f32>(f32(u.ext.x), f32(u.ext.y));
|
|
317
|
+
let uv = (vec2<f32>(px) + vec2<f32>(0.5)) / dims;
|
|
318
|
+
let ndc = vec4<f32>(uv.x * 2.0 - 1.0, 1.0 - uv.y * 2.0, depth, 1.0);
|
|
319
|
+
let w = u.inv_vp * ndc;
|
|
320
|
+
return w.xyz / w.w;
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
fn depth_at(px: vec2<i32>) -> f32 {
|
|
324
|
+
let clamped = clamp(px, vec2<i32>(0), vec2<i32>(i32(u.ext.x) - 1, i32(u.ext.y) - 1));
|
|
325
|
+
return textureLoad(depth_tex, clamped, 0);
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
fn is_sky(depth: f32) -> bool {
|
|
329
|
+
// Depth-buffer far plane. If the projection turns out reversed-Z the
|
|
330
|
+
// BLOOM_PT_DEBUG=1 depth view makes it obvious in one screenshot; flip
|
|
331
|
+
// here if geometry reads bright and sky reads dark.
|
|
332
|
+
return depth >= 0.9999999;
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
// Screen-space normal from depth: reconstruct neighbours, take the tighter
|
|
336
|
+
// derivative on each axis so depth discontinuities don't smear normals
|
|
337
|
+
// across silhouettes.
|
|
338
|
+
fn normal_from_depth(px: vec2<i32>, p_center: vec3<f32>) -> vec3<f32> {
|
|
339
|
+
let d_l = depth_at(px + vec2<i32>(-1, 0));
|
|
340
|
+
let d_r = depth_at(px + vec2<i32>(1, 0));
|
|
341
|
+
let d_u = depth_at(px + vec2<i32>(0, -1));
|
|
342
|
+
let d_d = depth_at(px + vec2<i32>(0, 1));
|
|
343
|
+
let d_c = depth_at(px);
|
|
344
|
+
|
|
345
|
+
var ddx: vec3<f32>;
|
|
346
|
+
if (abs(d_l - d_c) < abs(d_r - d_c)) {
|
|
347
|
+
ddx = p_center - world_at(px + vec2<i32>(-1, 0), d_l);
|
|
348
|
+
} else {
|
|
349
|
+
ddx = world_at(px + vec2<i32>(1, 0), d_r) - p_center;
|
|
350
|
+
}
|
|
351
|
+
var ddy: vec3<f32>;
|
|
352
|
+
if (abs(d_u - d_c) < abs(d_d - d_c)) {
|
|
353
|
+
ddy = p_center - world_at(px + vec2<i32>(0, -1), d_u);
|
|
354
|
+
} else {
|
|
355
|
+
ddy = world_at(px + vec2<i32>(0, 1), d_d) - p_center;
|
|
356
|
+
}
|
|
357
|
+
var n = cross(ddy, ddx);
|
|
358
|
+
let len = length(n);
|
|
359
|
+
if (len < 1e-8) { return vec3<f32>(0.0, 1.0, 0.0); }
|
|
360
|
+
n = n / len;
|
|
361
|
+
// Face the camera: a G-buffer surface always does.
|
|
362
|
+
if (dot(n, u.cam_pos.xyz - p_center) < 0.0) { n = -n; }
|
|
363
|
+
return n;
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
// ---- Sampling helpers ----------------------------------------------------------
|
|
367
|
+
|
|
368
|
+
// Branchless ONB (Duff et al. 2017).
|
|
369
|
+
fn onb(n: vec3<f32>) -> mat3x3<f32> {
|
|
370
|
+
let s = select(-1.0, 1.0, n.z >= 0.0);
|
|
371
|
+
let a = -1.0 / (s + n.z);
|
|
372
|
+
let b = n.x * n.y * a;
|
|
373
|
+
let t = vec3<f32>(1.0 + s * n.x * n.x * a, s * b, -s * n.x);
|
|
374
|
+
let bt = vec3<f32>(b, s + n.y * n.y * a, -n.y);
|
|
375
|
+
return mat3x3<f32>(t, bt, n);
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
fn cosine_sample(n: vec3<f32>, r: vec2<f32>) -> vec3<f32> {
|
|
379
|
+
let phi = 6.2831853 * r.x;
|
|
380
|
+
let sr = sqrt(r.y);
|
|
381
|
+
let local = vec3<f32>(cos(phi) * sr, sin(phi) * sr, sqrt(max(0.0, 1.0 - r.y)));
|
|
382
|
+
return normalize(onb(n) * local);
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
// Uniform direction in the solar cone (half-angle 0.265 deg -> soft shadows).
|
|
386
|
+
fn sun_cone_sample(r: vec2<f32>) -> vec3<f32> {
|
|
387
|
+
let cos_max = 0.9999893;
|
|
388
|
+
let cos_t = mix(cos_max, 1.0, r.x);
|
|
389
|
+
let sin_t = sqrt(max(0.0, 1.0 - cos_t * cos_t));
|
|
390
|
+
let phi = 6.2831853 * r.y;
|
|
391
|
+
let local = vec3<f32>(cos(phi) * sin_t, sin(phi) * sin_t, cos_t);
|
|
392
|
+
return normalize(onb(u.sun_dir.xyz) * local);
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
// Sky radiance for a miss. Analytic horizon-to-zenith gradient off the same
|
|
396
|
+
// ambient-derived tint Lumen's traces use; the sun disc is deliberately
|
|
397
|
+
// absent (the sun is sampled by NEE only, so it cannot be counted twice).
|
|
398
|
+
fn sky_radiance(dir: vec3<f32>) -> vec3<f32> {
|
|
399
|
+
let t = clamp(dir.y * 0.5 + 0.5, 0.0, 1.0);
|
|
400
|
+
return u.sky_color.rgb * mix(0.45, 1.35, t);
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
// ---- GGX BRDF sampling (PT-2; port of bloom-reference sample_brdf) --------
|
|
404
|
+
|
|
405
|
+
fn fresnel_schlick3(cos_theta: f32, f0: vec3<f32>) -> vec3<f32> {
|
|
406
|
+
let m = clamp(1.0 - cos_theta, 0.0, 1.0);
|
|
407
|
+
let m2 = m * m;
|
|
408
|
+
return f0 + (vec3<f32>(1.0) - f0) * (m2 * m2 * m);
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
fn smith_g1(n_dot_x: f32, alpha: f32) -> f32 {
|
|
412
|
+
let a2 = alpha * alpha;
|
|
413
|
+
let inner = sqrt((1.0 - a2) * n_dot_x * n_dot_x + a2);
|
|
414
|
+
return 2.0 * n_dot_x / (n_dot_x + inner + 1e-6);
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
fn v_smith(n_dot_v: f32, n_dot_l: f32, alpha: f32) -> f32 {
|
|
418
|
+
let a2 = alpha * alpha;
|
|
419
|
+
let ggx_v = n_dot_l * sqrt((n_dot_v * (1.0 - a2) + a2) * n_dot_v);
|
|
420
|
+
let ggx_l = n_dot_v * sqrt((n_dot_l * (1.0 - a2) + a2) * n_dot_l);
|
|
421
|
+
return 0.5 / (ggx_v + ggx_l + 1e-6);
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
fn burley_diffuse(n_dot_l: f32, n_dot_v: f32, l_dot_h: f32, roughness: f32) -> f32 {
|
|
425
|
+
let fd90 = 0.5 + 2.0 * l_dot_h * l_dot_h * roughness;
|
|
426
|
+
let ml = pow(1.0 - n_dot_l, 5.0);
|
|
427
|
+
let mv = pow(1.0 - n_dot_v, 5.0);
|
|
428
|
+
return (1.0 + (fd90 - 1.0) * ml) * (1.0 + (fd90 - 1.0) * mv) / 3.14159265;
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
// Heitz 2018 VNDF sampler — visible-normal distribution, tangent frame.
|
|
432
|
+
fn sample_ggx_vndf(v_t: vec3<f32>, alpha: f32, r2: vec2<f32>) -> vec3<f32> {
|
|
433
|
+
let vh = normalize(vec3<f32>(alpha * v_t.x, alpha * v_t.y, v_t.z));
|
|
434
|
+
let lensq = vh.x * vh.x + vh.y * vh.y;
|
|
435
|
+
var t1 = vec3<f32>(1.0, 0.0, 0.0);
|
|
436
|
+
if (lensq > 0.0) {
|
|
437
|
+
t1 = vec3<f32>(-vh.y, vh.x, 0.0) / sqrt(lensq);
|
|
438
|
+
}
|
|
439
|
+
let t2 = cross(vh, t1);
|
|
440
|
+
let r = sqrt(r2.x);
|
|
441
|
+
let phi = 6.2831853 * r2.y;
|
|
442
|
+
let t1v = r * cos(phi);
|
|
443
|
+
var t2v = r * sin(phi);
|
|
444
|
+
let s = 0.5 * (1.0 + vh.z);
|
|
445
|
+
t2v = (1.0 - s) * sqrt(max(0.0, 1.0 - t1v * t1v)) + s * t2v;
|
|
446
|
+
let nh = t1v * t1 + t2v * t2 + sqrt(max(0.0, 1.0 - t1v * t1v - t2v * t2v)) * vh;
|
|
447
|
+
return normalize(vec3<f32>(alpha * nh.x, alpha * nh.y, max(nh.z, 0.0)));
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
struct BrdfSample {
|
|
451
|
+
dir: vec3<f32>,
|
|
452
|
+
// BRDF * cos / pdf, physical convention. For the pure-diffuse case
|
|
453
|
+
// this reduces to plain albedo, so the game's pi-premultiplied
|
|
454
|
+
// light intensities are unaffected.
|
|
455
|
+
weight: vec3<f32>,
|
|
456
|
+
valid: bool,
|
|
457
|
+
};
|
|
458
|
+
|
|
459
|
+
fn sample_brdf(
|
|
460
|
+
n: vec3<f32>,
|
|
461
|
+
view_ws: vec3<f32>,
|
|
462
|
+
base_color: vec3<f32>,
|
|
463
|
+
roughness: f32,
|
|
464
|
+
metallic: f32,
|
|
465
|
+
) -> BrdfSample {
|
|
466
|
+
var out: BrdfSample;
|
|
467
|
+
out.valid = false;
|
|
468
|
+
let alpha = max(roughness * roughness, 1e-3);
|
|
469
|
+
let m = onb(n); // columns (t, bt, n): local -> world
|
|
470
|
+
let v_t = vec3<f32>(dot(view_ws, m[0]), dot(view_ws, m[1]), dot(view_ws, n));
|
|
471
|
+
if (v_t.z <= 0.0) {
|
|
472
|
+
return out;
|
|
473
|
+
}
|
|
474
|
+
let f0 = mix(vec3<f32>(0.04), base_color, metallic);
|
|
475
|
+
// Lobe pick by Fresnel at the ACTUAL view angle, not at normal
|
|
476
|
+
// incidence: at grazing angles specular energy approaches 1, and
|
|
477
|
+
// the estimator divides by the pick probability — picking with the
|
|
478
|
+
// ~0.04 normal-incidence weight amplified rare grazing specular
|
|
479
|
+
// samples ~25x into a field of white fireflies at 1-2 spp (the
|
|
480
|
+
// whole ground plane is grazing at distance). Clamped so neither
|
|
481
|
+
// lobe's 1/p boost can exceed ~20x even in edge cases.
|
|
482
|
+
let n_dot_v_pick = max(dot(n, view_ws), 0.0);
|
|
483
|
+
let f_view = fresnel_schlick3(n_dot_v_pick, f0);
|
|
484
|
+
let spec_weight = (f_view.x + f_view.y + f_view.z) / 3.0;
|
|
485
|
+
let diff_weight = (1.0 - spec_weight) * (1.0 - metallic);
|
|
486
|
+
var p_spec = spec_weight / (spec_weight + diff_weight + 1e-6);
|
|
487
|
+
p_spec = clamp(p_spec, 0.05, 0.95);
|
|
488
|
+
let r2 = rand_2f();
|
|
489
|
+
if (rand_f() < p_spec) {
|
|
490
|
+
let h_t = sample_ggx_vndf(v_t, alpha, r2);
|
|
491
|
+
let l_t = reflect(-v_t, h_t);
|
|
492
|
+
if (l_t.z <= 0.0) {
|
|
493
|
+
return out;
|
|
494
|
+
}
|
|
495
|
+
let n_dot_l = l_t.z;
|
|
496
|
+
let n_dot_v = max(v_t.z, 1e-4);
|
|
497
|
+
let v_dot_h = max(dot(v_t, h_t), 1e-4);
|
|
498
|
+
let f = fresnel_schlick3(v_dot_h, f0);
|
|
499
|
+
// VNDF pdf: throughput collapses to F * G2 / G1(V).
|
|
500
|
+
let g2 = v_smith(n_dot_v, n_dot_l, alpha) * 4.0 * n_dot_v * n_dot_l;
|
|
501
|
+
let g1_v = smith_g1(n_dot_v, alpha);
|
|
502
|
+
out.dir = m * l_t;
|
|
503
|
+
out.weight = f * g2 / (max(g1_v, 1e-6) * p_spec);
|
|
504
|
+
// Realtime mode trades a little energy for stability: a single
|
|
505
|
+
// bounce may not multiply throughput more than 4x (the ~7-frame
|
|
506
|
+
// EMA window cannot average outliers away like progressive
|
|
507
|
+
// accumulation can). Progressive mode stays unclamped.
|
|
508
|
+
if (u.cfg.x >= 2.0) {
|
|
509
|
+
out.weight = min(out.weight, vec3<f32>(4.0));
|
|
510
|
+
}
|
|
511
|
+
out.valid = true;
|
|
512
|
+
return out;
|
|
513
|
+
}
|
|
514
|
+
// Diffuse lobe: cosine hemisphere; weight = albedo * burley * pi
|
|
515
|
+
// (Burley divides by pi internally; pdf = cos/pi cancels the cos).
|
|
516
|
+
let r = sqrt(r2.x);
|
|
517
|
+
let phi = 6.2831853 * r2.y;
|
|
518
|
+
let l_t = vec3<f32>(r * cos(phi), r * sin(phi), sqrt(max(0.0, 1.0 - r2.x)));
|
|
519
|
+
let n_dot_l = max(l_t.z, 1e-4);
|
|
520
|
+
let n_dot_v = max(v_t.z, 1e-4);
|
|
521
|
+
let h_un = v_t + l_t;
|
|
522
|
+
var l_dot_h = 0.0;
|
|
523
|
+
if (dot(h_un, h_un) > 1e-8) {
|
|
524
|
+
l_dot_h = max(dot(l_t, normalize(h_un)), 0.0);
|
|
525
|
+
}
|
|
526
|
+
let diffuse_albedo = base_color * (1.0 - metallic) * (vec3<f32>(1.0) - f0);
|
|
527
|
+
let fd = burley_diffuse(n_dot_l, n_dot_v, l_dot_h, roughness);
|
|
528
|
+
out.dir = m * l_t;
|
|
529
|
+
out.weight = diffuse_albedo * fd * 3.14159265 / (1.0 - p_spec);
|
|
530
|
+
if (u.cfg.x >= 2.0) {
|
|
531
|
+
out.weight = min(out.weight, vec3<f32>(4.0));
|
|
532
|
+
}
|
|
533
|
+
out.valid = true;
|
|
534
|
+
return out;
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
// ---- Ray casts ------------------------------------------------------------------
|
|
538
|
+
|
|
539
|
+
fn occluded(origin: vec3<f32>, dir: vec3<f32>, max_t: f32) -> bool {
|
|
540
|
+
var rq: ray_query;
|
|
541
|
+
rayQueryInitialize(&rq, accel, RayDesc(0u, 0xFFu, 0.001, max_t, origin, dir));
|
|
542
|
+
loop {
|
|
543
|
+
if (!rayQueryProceed(&rq)) { break; }
|
|
544
|
+
}
|
|
545
|
+
let hit = rayQueryGetCommittedIntersection(&rq);
|
|
546
|
+
return hit.kind != RAY_QUERY_INTERSECTION_NONE;
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
// ---- Hit shading: card albedo -----------------------------------------------------
|
|
550
|
+
|
|
551
|
+
// Same signed-axis card projection as the Lumen HW trace (ssgi.rs), but
|
|
552
|
+
// sampling the raw ALBEDO atlas. Falls back to the flat instance albedo when
|
|
553
|
+
// the mesh has no captured card.
|
|
554
|
+
fn albedo_at_hit(
|
|
555
|
+
inst: InstanceGiData,
|
|
556
|
+
hit_os: vec3<f32>,
|
|
557
|
+
dir_ws: vec3<f32>,
|
|
558
|
+
) -> vec3<f32> {
|
|
559
|
+
if (inst.card_slot.w <= 0.5) {
|
|
560
|
+
return inst.albedo;
|
|
561
|
+
}
|
|
562
|
+
let abs_d = abs(dir_ws);
|
|
563
|
+
var axis_idx: u32 = 0u;
|
|
564
|
+
if (abs_d.y >= abs_d.x && abs_d.y >= abs_d.z) {
|
|
565
|
+
axis_idx = 2u;
|
|
566
|
+
} else if (abs_d.z >= abs_d.x) {
|
|
567
|
+
axis_idx = 4u;
|
|
568
|
+
}
|
|
569
|
+
var signed_axis: u32 = axis_idx;
|
|
570
|
+
if (axis_idx == 0u && dir_ws.x > 0.0) { signed_axis = 1u; }
|
|
571
|
+
else if (axis_idx == 2u && dir_ws.y > 0.0) { signed_axis = 3u; }
|
|
572
|
+
else if (axis_idx == 4u && dir_ws.z > 0.0) { signed_axis = 5u; }
|
|
573
|
+
|
|
574
|
+
let first_slot = u32(inst.card_slot.x);
|
|
575
|
+
let slot = first_slot + signed_axis;
|
|
576
|
+
let slot_x = slot % 64u;
|
|
577
|
+
let slot_y = slot / 64u;
|
|
578
|
+
|
|
579
|
+
let bmin = inst.card_aabb_min.xyz;
|
|
580
|
+
let bmax = inst.card_aabb_max.xyz;
|
|
581
|
+
var u_os: f32; var v_os: f32;
|
|
582
|
+
var u_lo: f32; var u_hi: f32; var v_lo: f32; var v_hi: f32;
|
|
583
|
+
var u_flip: f32 = 1.0;
|
|
584
|
+
if (signed_axis == 0u || signed_axis == 1u) {
|
|
585
|
+
u_os = hit_os.y; v_os = hit_os.z;
|
|
586
|
+
u_lo = bmin.y; u_hi = bmax.y; v_lo = bmin.z; v_hi = bmax.z;
|
|
587
|
+
if (signed_axis == 1u) { u_flip = -1.0; }
|
|
588
|
+
} else if (signed_axis == 2u || signed_axis == 3u) {
|
|
589
|
+
u_os = hit_os.x; v_os = hit_os.z;
|
|
590
|
+
u_lo = bmin.x; u_hi = bmax.x; v_lo = bmin.z; v_hi = bmax.z;
|
|
591
|
+
if (signed_axis == 3u) { u_flip = -1.0; }
|
|
592
|
+
} else {
|
|
593
|
+
u_os = hit_os.x; v_os = hit_os.y;
|
|
594
|
+
u_lo = bmin.x; u_hi = bmax.x; v_lo = bmin.y; v_hi = bmax.y;
|
|
595
|
+
if (signed_axis == 5u) { u_flip = -1.0; }
|
|
596
|
+
}
|
|
597
|
+
var u_norm = clamp((u_os - u_lo) / max(u_hi - u_lo, 1e-4), 0.0, 1.0);
|
|
598
|
+
let v_norm = clamp((v_os - v_lo) / max(v_hi - v_lo, 1e-4), 0.0, 1.0);
|
|
599
|
+
if (u_flip < 0.0) { u_norm = 1.0 - u_norm; }
|
|
600
|
+
|
|
601
|
+
let slot_size_uv = 1.0 / 64.0;
|
|
602
|
+
let texel_in_slot = slot_size_uv / 64.0;
|
|
603
|
+
let slot_u0 = f32(slot_x) * slot_size_uv + texel_in_slot;
|
|
604
|
+
let slot_v0 = f32(slot_y) * slot_size_uv + texel_in_slot;
|
|
605
|
+
let slot_span = slot_size_uv - 2.0 * texel_in_slot;
|
|
606
|
+
let atlas_uv = vec2<f32>(slot_u0 + u_norm * slot_span, slot_v0 + v_norm * slot_span);
|
|
607
|
+
return textureSampleLevel(card_albedo_atlas, card_samp, atlas_uv, 0.0).rgb;
|
|
608
|
+
}
|
|
609
|
+
|
|
610
|
+
// ---- Next-event estimation ---------------------------------------------------------
|
|
611
|
+
|
|
612
|
+
// Direct light at a surface point: sun through the solar cone + one point
|
|
613
|
+
// light chosen uniformly (contribution / pdf). Game-radiometry convention:
|
|
614
|
+
// no 1/pi (see file header).
|
|
615
|
+
// Sun visibility at a surface point: shadow cascades in hybrid mode
|
|
616
|
+
// (deterministic, matches the raster shadows exactly), a traced cone
|
|
617
|
+
// ray otherwise (reference quality, soft penumbra).
|
|
618
|
+
fn sun_visibility(p: vec3<f32>, n: vec3<f32>, r2: vec2<f32>) -> f32 {
|
|
619
|
+
if (u.ext.z == 1u) {
|
|
620
|
+
return sun_vis_cascade(p);
|
|
621
|
+
}
|
|
622
|
+
let sd = sun_cone_sample(r2);
|
|
623
|
+
if (dot(n, sd) <= 0.0) {
|
|
624
|
+
return 0.0;
|
|
625
|
+
}
|
|
626
|
+
if (occluded(p, sd, 1000.0)) {
|
|
627
|
+
return 0.0;
|
|
628
|
+
}
|
|
629
|
+
return 1.0;
|
|
630
|
+
}
|
|
631
|
+
|
|
632
|
+
// GGX highlight for an NEE light sample — the same D/F/V terms as
|
|
633
|
+
// sample_brdf, evaluated for a known light direction. The shadow ray is
|
|
634
|
+
// already paid for by the diffuse term, so specular NEE rides along
|
|
635
|
+
// free (PT-5: point lights and bounce vertices were diffuse-only, the
|
|
636
|
+
// documented PT-2 gap). Analytic lights cannot be hit by BSDF rays and
|
|
637
|
+
// sky misses exclude the sun disc, so nothing double-counts.
|
|
638
|
+
fn nee_spec(n: vec3<f32>, view: vec3<f32>, ldir: vec3<f32>, ndl: f32,
|
|
639
|
+
full_alb: vec3<f32>, rough: f32, metal: f32) -> vec3<f32> {
|
|
640
|
+
let hv = normalize(view + ldir);
|
|
641
|
+
let ndv = max(dot(n, view), 1e-4);
|
|
642
|
+
let ndh = max(dot(n, hv), 0.0);
|
|
643
|
+
let vdh = max(dot(view, hv), 1e-4);
|
|
644
|
+
let alpha0 = max(rough * rough, 1e-3);
|
|
645
|
+
let a2 = alpha0 * alpha0;
|
|
646
|
+
let dd = ndh * ndh * (a2 - 1.0) + 1.0;
|
|
647
|
+
let dterm = a2 / (3.14159265 * dd * dd);
|
|
648
|
+
let f0s = mix(vec3<f32>(0.04), full_alb, metal);
|
|
649
|
+
return fresnel_schlick3(vdh, f0s) * dterm * v_smith(ndv, ndl, alpha0) * ndl;
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
fn direct_light(p: vec3<f32>, n: vec3<f32>, alb: vec3<f32>, sun_r2: vec2<f32>,
|
|
653
|
+
view: vec3<f32>, full_alb: vec3<f32>, rough: f32, metal: f32,
|
|
654
|
+
with_points: bool) -> vec3<f32> {
|
|
655
|
+
var lit = vec3<f32>(0.0);
|
|
656
|
+
var spec = vec3<f32>(0.0);
|
|
657
|
+
|
|
658
|
+
let ndl = max(dot(n, u.sun_dir.xyz), 0.0);
|
|
659
|
+
if (ndl > 0.0) {
|
|
660
|
+
let vis = sun_visibility(p, n, sun_r2);
|
|
661
|
+
lit += u.sun_color.rgb * ndl * vis;
|
|
662
|
+
if (vis > 0.0) {
|
|
663
|
+
spec += nee_spec(n, view, u.sun_dir.xyz, ndl, full_alb, rough, metal)
|
|
664
|
+
* u.sun_color.rgb * vis;
|
|
665
|
+
}
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
let count = u32(u.cfg.z);
|
|
669
|
+
if (count > 0u && with_points) {
|
|
670
|
+
let pick = min(u32(rand_f() * f32(count)), count - 1u);
|
|
671
|
+
let l = u.lights[pick];
|
|
672
|
+
let to_l = l.pos_range.xyz - p;
|
|
673
|
+
let d = length(to_l);
|
|
674
|
+
let range = l.pos_range.w;
|
|
675
|
+
if (d < range && d > 1e-3) {
|
|
676
|
+
let dir = to_l / d;
|
|
677
|
+
let ndl2 = dot(n, dir);
|
|
678
|
+
if (ndl2 > 0.0 && !occluded(p, dir, d - 0.02)) {
|
|
679
|
+
// Raster-parity falloff: (1 - d/range)^2, core.rs.
|
|
680
|
+
let att = 1.0 - d / range;
|
|
681
|
+
let li = l.color_int.rgb * l.color_int.w * att * att * f32(count);
|
|
682
|
+
lit += li * ndl2;
|
|
683
|
+
spec += nee_spec(n, view, dir, ndl2, full_alb, rough, metal) * li;
|
|
684
|
+
}
|
|
685
|
+
}
|
|
686
|
+
}
|
|
687
|
+
return alb * lit + spec;
|
|
688
|
+
}
|
|
689
|
+
|
|
690
|
+
// ---- PT-4 (EXPERIMENTAL) — ReSTIR DI over the analytic point lights -------
|
|
691
|
+
//
|
|
692
|
+
// RIS with 8 uniform candidates + temporal reservoir reuse (M-capped at
|
|
693
|
+
// 20x), one shadow ray for the winner. The target is re-evaluated at
|
|
694
|
+
// the CURRENT shading point when merging history, so the temporal reuse
|
|
695
|
+
// carries no geometric bias; visibility reuse bias does not arise
|
|
696
|
+
// because visibility is never folded into the reservoir. With this
|
|
697
|
+
// game's <=16 analytic lights plain NEE is nearly as good (the roadmap
|
|
698
|
+
// said so up front) — this lands the architecture for the day emissive
|
|
699
|
+
// particles/muzzle flashes become real light sources.
|
|
700
|
+
|
|
701
|
+
// Unshadowed contribution of light `li` at the shading point.
|
|
702
|
+
fn restir_contrib(li: u32, p: vec3<f32>, n: vec3<f32>, view: vec3<f32>,
|
|
703
|
+
alb_diff: vec3<f32>, full_alb: vec3<f32>,
|
|
704
|
+
rough: f32, metal: f32) -> vec3<f32> {
|
|
705
|
+
let l = u.lights[li];
|
|
706
|
+
let to_l = l.pos_range.xyz - p;
|
|
707
|
+
let d = length(to_l);
|
|
708
|
+
let range = l.pos_range.w;
|
|
709
|
+
if (d >= range || d <= 1e-3) { return vec3<f32>(0.0); }
|
|
710
|
+
let dir = to_l / d;
|
|
711
|
+
let ndl = dot(n, dir);
|
|
712
|
+
if (ndl <= 0.0) { return vec3<f32>(0.0); }
|
|
713
|
+
let att = 1.0 - d / range;
|
|
714
|
+
let li_rgb = l.color_int.rgb * l.color_int.w * att * att;
|
|
715
|
+
return alb_diff * li_rgb * ndl
|
|
716
|
+
+ nee_spec(n, view, dir, ndl, full_alb, rough, metal) * li_rgb;
|
|
717
|
+
}
|
|
718
|
+
|
|
719
|
+
// Scalar target density: luminance of the unshadowed contribution.
|
|
720
|
+
// Correctness never depends on the target — only variance does.
|
|
721
|
+
fn restir_target(li: u32, p: vec3<f32>, n: vec3<f32>, view: vec3<f32>,
|
|
722
|
+
alb_diff: vec3<f32>, full_alb: vec3<f32>,
|
|
723
|
+
rough: f32, metal: f32) -> f32 {
|
|
724
|
+
return dot(restir_contrib(li, p, n, view, alb_diff, full_alb, rough, metal),
|
|
725
|
+
vec3<f32>(0.2126, 0.7152, 0.0722));
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
// Runs at the primary vertex when ext.w == 1; writes this frame's
|
|
729
|
+
// reservoir and returns the winner's shadow-tested contribution.
|
|
730
|
+
fn restir_point_light(idx: u32, p: vec3<f32>, n: vec3<f32>, view: vec3<f32>,
|
|
731
|
+
alb_diff: vec3<f32>, full_alb: vec3<f32>,
|
|
732
|
+
rough: f32, metal: f32) -> vec3<f32> {
|
|
733
|
+
let count = u32(u.cfg.z);
|
|
734
|
+
if (count == 0u) {
|
|
735
|
+
resv_out[idx] = vec4<f32>(-1.0, 0.0, 0.0, 0.0);
|
|
736
|
+
return vec3<f32>(0.0);
|
|
737
|
+
}
|
|
738
|
+
var r_y = 0u;
|
|
739
|
+
var r_wsum = 0.0;
|
|
740
|
+
var r_m = 0.0;
|
|
741
|
+
var r_phat = 0.0;
|
|
742
|
+
// RIS: 8 uniform candidates (pdf = 1/count => w = phat * count).
|
|
743
|
+
for (var c = 0u; c < 8u; c = c + 1u) {
|
|
744
|
+
let cand = min(u32(rand_f() * f32(count)), count - 1u);
|
|
745
|
+
let ph = restir_target(cand, p, n, view, alb_diff, full_alb, rough, metal);
|
|
746
|
+
let w = ph * f32(count);
|
|
747
|
+
r_wsum += w;
|
|
748
|
+
r_m += 1.0;
|
|
749
|
+
if (w > 0.0 && rand_f() * r_wsum < w) {
|
|
750
|
+
r_y = cand;
|
|
751
|
+
r_phat = ph;
|
|
752
|
+
}
|
|
753
|
+
}
|
|
754
|
+
// Temporal reuse from the reprojected texel. The stored W already
|
|
755
|
+
// integrates that reservoir's history; its M is capped so stale
|
|
756
|
+
// samples cannot outvote fresh ones forever.
|
|
757
|
+
if (rp_valid) {
|
|
758
|
+
let pr = resv[rp_nearest];
|
|
759
|
+
let pm = min(pr.z, 160.0);
|
|
760
|
+
if (pm > 0.0 && pr.x >= 0.0 && u32(pr.x) < count) {
|
|
761
|
+
let py = u32(pr.x);
|
|
762
|
+
let ph = restir_target(py, p, n, view, alb_diff, full_alb, rough, metal);
|
|
763
|
+
let w = ph * pr.y * pm;
|
|
764
|
+
if (w > 0.0) {
|
|
765
|
+
r_wsum += w;
|
|
766
|
+
if (rand_f() * r_wsum < w) {
|
|
767
|
+
r_y = py;
|
|
768
|
+
r_phat = ph;
|
|
769
|
+
}
|
|
770
|
+
}
|
|
771
|
+
r_m += pm;
|
|
772
|
+
}
|
|
773
|
+
}
|
|
774
|
+
var r_w = 0.0;
|
|
775
|
+
if (r_phat > 0.0 && r_m > 0.0) {
|
|
776
|
+
r_w = r_wsum / (r_m * r_phat);
|
|
777
|
+
}
|
|
778
|
+
resv_out[idx] = vec4<f32>(f32(r_y), r_w, r_m, r_phat);
|
|
779
|
+
if (r_w <= 0.0) { return vec3<f32>(0.0); }
|
|
780
|
+
// One shadow ray for the winner.
|
|
781
|
+
let l = u.lights[r_y];
|
|
782
|
+
let to_l = l.pos_range.xyz - p;
|
|
783
|
+
let d = length(to_l);
|
|
784
|
+
if (d <= 1e-3) { return vec3<f32>(0.0); }
|
|
785
|
+
let dir = to_l / d;
|
|
786
|
+
if (occluded(p, dir, d - 0.02)) { return vec3<f32>(0.0); }
|
|
787
|
+
return restir_contrib(r_y, p, n, view, alb_diff, full_alb, rough, metal) * r_w;
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
// ---- Main -----------------------------------------------------------------------------
|
|
791
|
+
|
|
792
|
+
@compute @workgroup_size(8, 8, 1)
|
|
793
|
+
fn cs_main(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
794
|
+
if (gid.x >= u.size.x || gid.y >= u.size.y) { return; }
|
|
795
|
+
let px = vec2<i32>(i32(gid.x), i32(gid.y));
|
|
796
|
+
// A rolling per-frame sequence in EVERY mode: SVGF's temporal
|
|
797
|
+
// accumulation needs unbiased fresh samples each frame (a frozen
|
|
798
|
+
// sequence converges the EMA to a wrong-but-stable value that
|
|
799
|
+
// reads as static dirty-lens grain glued to the screen). The
|
|
800
|
+
// variance estimate below is what keeps rolling noise from
|
|
801
|
+
// shimmering: it tells the à-trous exactly where to blur hard.
|
|
802
|
+
rng_seed(gid.xy, u.size.z);
|
|
803
|
+
|
|
804
|
+
// PT-3 half-res: realtime mode traces a half grid; map this trace
|
|
805
|
+
// cell to its full-res G-buffer pixel (the 2x2 phase rotates per
|
|
806
|
+
// frame so the EMA integrates all four over time). Full-res modes
|
|
807
|
+
// have ext == size and phase 0, making this the identity.
|
|
808
|
+
var px_full = px;
|
|
809
|
+
if (u.ext.x > u.size.x) {
|
|
810
|
+
// Generalized ratio (the trace grid is budget-capped, so the
|
|
811
|
+
// factor is not necessarily 2): integer scale keeps each trace
|
|
812
|
+
// texel pinned to one owner pixel.
|
|
813
|
+
px_full = min(
|
|
814
|
+
vec2<i32>(
|
|
815
|
+
px.x * i32(u.ext.x) / i32(u.size.x),
|
|
816
|
+
px.y * i32(u.ext.y) / i32(u.size.y),
|
|
817
|
+
),
|
|
818
|
+
vec2<i32>(i32(u.ext.x) - 1, i32(u.ext.y) - 1),
|
|
819
|
+
);
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
let debug = u.cfg.w;
|
|
823
|
+
if (debug == 5.0) {
|
|
824
|
+
textureStore(out_hdr, px_full, vec4<f32>(1.0, 0.0, 1.0, 1.0));
|
|
825
|
+
return;
|
|
826
|
+
}
|
|
827
|
+
|
|
828
|
+
let depth = depth_at(px_full);
|
|
829
|
+
if (debug == 1.0) {
|
|
830
|
+
textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(depth), 1.0));
|
|
831
|
+
return;
|
|
832
|
+
}
|
|
833
|
+
|
|
834
|
+
if (is_sky(depth)) {
|
|
835
|
+
// Leave the raster sky/clouds untouched. Realtime mode marks
|
|
836
|
+
// the texel as sky in the MOMENTS buffer (depth channel = far
|
|
837
|
+
// plane) so the a-trous passes and the upsampler skip it.
|
|
838
|
+
if (u.cfg.x >= 2.0 && u.cfg.w == 0.0) {
|
|
839
|
+
let sky_idx = gid.y * u.size.x + gid.x;
|
|
840
|
+
accum_out[sky_idx] = vec4<f32>(0.0);
|
|
841
|
+
moments_out[sky_idx] = vec4<f32>(0.0, 0.0, 0.0, 1.0);
|
|
842
|
+
}
|
|
843
|
+
return;
|
|
844
|
+
}
|
|
845
|
+
|
|
846
|
+
let p0 = world_at(px_full, depth);
|
|
847
|
+
let n0 = normal_from_depth(px_full, p0);
|
|
848
|
+
let albedo0 = textureLoad(albedo_tex, px_full, 0).rgb;
|
|
849
|
+
|
|
850
|
+
if (debug == 2.0) {
|
|
851
|
+
textureStore(out_hdr, px_full, vec4<f32>(n0 * 0.5 + 0.5, 1.0));
|
|
852
|
+
return;
|
|
853
|
+
}
|
|
854
|
+
if (debug == 3.0) {
|
|
855
|
+
textureStore(out_hdr, px_full, vec4<f32>(albedo0, 1.0));
|
|
856
|
+
return;
|
|
857
|
+
}
|
|
858
|
+
if (debug == 4.0) {
|
|
859
|
+
let sd = sun_cone_sample(rand_2f());
|
|
860
|
+
let vis = select(0.0, 1.0, !occluded(p0 + n0 * 0.02, sd, 1000.0));
|
|
861
|
+
textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(vis), 1.0));
|
|
862
|
+
return;
|
|
863
|
+
}
|
|
864
|
+
if (debug == 8.0) {
|
|
865
|
+
// Binary probe: white = traced hit has a geometry window,
|
|
866
|
+
// black = geo.z reads 0, red = TLAS miss. HDR-large values so
|
|
867
|
+
// exposure/tonemap can't blur the verdict.
|
|
868
|
+
let dir0 = normalize(p0 - u.cam_pos.xyz);
|
|
869
|
+
var rq8: ray_query;
|
|
870
|
+
rayQueryInitialize(&rq8, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
|
|
871
|
+
loop {
|
|
872
|
+
if (!rayQueryProceed(&rq8)) { break; }
|
|
873
|
+
}
|
|
874
|
+
let h8 = rayQueryGetCommittedIntersection(&rq8);
|
|
875
|
+
var c8 = vec3<f32>(100.0, 0.0, 0.0);
|
|
876
|
+
if (h8.kind != RAY_QUERY_INTERSECTION_NONE) {
|
|
877
|
+
let gi = instance_data[h8.instance_custom_data].geo;
|
|
878
|
+
c8 = select(vec3<f32>(0.0), vec3<f32>(100.0), gi.z > 0u);
|
|
879
|
+
}
|
|
880
|
+
textureStore(out_hdr, px_full, vec4<f32>(c8, 1.0));
|
|
881
|
+
return;
|
|
882
|
+
}
|
|
883
|
+
if (debug == 9.0) {
|
|
884
|
+
// Quantized normal probe: dominant axis of the interpolated
|
|
885
|
+
// world normal as six saturated HDR colours. +X red, -X dark
|
|
886
|
+
// red-ish magenta, +Y green, -Y cyan, +Z blue, -Z yellow.
|
|
887
|
+
// Gray = TLAS miss / no window / zero-length normal.
|
|
888
|
+
let dir0 = normalize(p0 - u.cam_pos.xyz);
|
|
889
|
+
var rq9: ray_query;
|
|
890
|
+
rayQueryInitialize(&rq9, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
|
|
891
|
+
loop {
|
|
892
|
+
if (!rayQueryProceed(&rq9)) { break; }
|
|
893
|
+
}
|
|
894
|
+
let h9 = rayQueryGetCommittedIntersection(&rq9);
|
|
895
|
+
var c9 = vec3<f32>(5.0, 5.0, 5.0);
|
|
896
|
+
if (h9.kind != RAY_QUERY_INTERSECTION_NONE) {
|
|
897
|
+
let inst9 = instance_data[h9.instance_custom_data];
|
|
898
|
+
if (inst9.geo.z > 0u) {
|
|
899
|
+
let a9 = fetch_hit_attrs(inst9.geo, h9.primitive_index, h9.barycentrics);
|
|
900
|
+
let raw = a9.normal_os;
|
|
901
|
+
if (length(raw) > 1e-6) {
|
|
902
|
+
let n9 = normal_to_world(raw, h9.world_to_object);
|
|
903
|
+
let an = abs(n9);
|
|
904
|
+
if (an.y >= an.x && an.y >= an.z) {
|
|
905
|
+
c9 = select(vec3<f32>(0.0, 50.0, 50.0), vec3<f32>(0.0, 50.0, 0.0), n9.y >= 0.0);
|
|
906
|
+
} else if (an.x >= an.z) {
|
|
907
|
+
c9 = select(vec3<f32>(50.0, 0.0, 25.0), vec3<f32>(50.0, 0.0, 0.0), n9.x >= 0.0);
|
|
908
|
+
} else {
|
|
909
|
+
c9 = select(vec3<f32>(50.0, 50.0, 0.0), vec3<f32>(0.0, 0.0, 50.0), n9.z >= 0.0);
|
|
910
|
+
}
|
|
911
|
+
}
|
|
912
|
+
}
|
|
913
|
+
}
|
|
914
|
+
textureStore(out_hdr, px_full, vec4<f32>(c9, 1.0));
|
|
915
|
+
return;
|
|
916
|
+
}
|
|
917
|
+
if (debug == 10.0) {
|
|
918
|
+
// primitive_index sanity probe: banded pseudo-colour of the hit
|
|
919
|
+
// triangle index. Expected: per-triangle colour noise across
|
|
920
|
+
// meshes. A single flat colour everywhere = the field is
|
|
921
|
+
// constant; saturated white = garbage-huge.
|
|
922
|
+
let dir0 = normalize(p0 - u.cam_pos.xyz);
|
|
923
|
+
var rq10: ray_query;
|
|
924
|
+
rayQueryInitialize(&rq10, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
|
|
925
|
+
loop {
|
|
926
|
+
if (!rayQueryProceed(&rq10)) { break; }
|
|
927
|
+
}
|
|
928
|
+
let h10 = rayQueryGetCommittedIntersection(&rq10);
|
|
929
|
+
var c10 = vec3<f32>(0.0);
|
|
930
|
+
if (h10.kind != RAY_QUERY_INTERSECTION_NONE) {
|
|
931
|
+
let prim = f32(h10.primitive_index);
|
|
932
|
+
c10 = vec3<f32>(fract(prim / 64.0), fract(prim / 1024.0), fract(prim / 16384.0)) * 30.0;
|
|
933
|
+
}
|
|
934
|
+
textureStore(out_hdr, px_full, vec4<f32>(c10, 1.0));
|
|
935
|
+
return;
|
|
936
|
+
}
|
|
937
|
+
if (debug == 11.0 || debug == 12.0) {
|
|
938
|
+
// 11: instance_custom_data palette (expect distinct colours per
|
|
939
|
+
// proxy: terrain vs trees vs building). Constant = broken.
|
|
940
|
+
// 12: raw barycentrics (expect smooth per-triangle gradients).
|
|
941
|
+
let dir0 = normalize(p0 - u.cam_pos.xyz);
|
|
942
|
+
var rq11: ray_query;
|
|
943
|
+
rayQueryInitialize(&rq11, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
|
|
944
|
+
loop {
|
|
945
|
+
if (!rayQueryProceed(&rq11)) { break; }
|
|
946
|
+
}
|
|
947
|
+
let h11 = rayQueryGetCommittedIntersection(&rq11);
|
|
948
|
+
var c11 = vec3<f32>(0.0);
|
|
949
|
+
if (h11.kind != RAY_QUERY_INTERSECTION_NONE) {
|
|
950
|
+
if (debug == 11.0) {
|
|
951
|
+
let id = h11.instance_custom_data;
|
|
952
|
+
c11 = vec3<f32>(
|
|
953
|
+
f32((id * 37u) % 7u) / 7.0,
|
|
954
|
+
f32((id * 61u) % 11u) / 11.0,
|
|
955
|
+
f32((id * 13u) % 5u) / 5.0,
|
|
956
|
+
) * 30.0;
|
|
957
|
+
} else {
|
|
958
|
+
let b = h11.barycentrics;
|
|
959
|
+
c11 = vec3<f32>(b.x, b.y, max(0.0, 1.0 - b.x - b.y)) * 30.0;
|
|
960
|
+
}
|
|
961
|
+
}
|
|
962
|
+
textureStore(out_hdr, px_full, vec4<f32>(c11, 1.0));
|
|
963
|
+
return;
|
|
964
|
+
}
|
|
965
|
+
if (debug == 13.0) {
|
|
966
|
+
// TLAS sanity: green = traced primary hit distance agrees with
|
|
967
|
+
// the G-buffer depth (within 2% + 0.1m), red = disagreement
|
|
968
|
+
// (wrong geometry committed), blue = TLAS miss on a G-buffer
|
|
969
|
+
// pixel. If this is red/blue everywhere the TLAS itself (not
|
|
970
|
+
// the intersection attributes) is broken on this backend.
|
|
971
|
+
let to_p = p0 - u.cam_pos.xyz;
|
|
972
|
+
let gdist = length(to_p);
|
|
973
|
+
let dir0 = to_p / max(gdist, 1e-4);
|
|
974
|
+
var rq13: ray_query;
|
|
975
|
+
rayQueryInitialize(&rq13, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
|
|
976
|
+
loop {
|
|
977
|
+
if (!rayQueryProceed(&rq13)) { break; }
|
|
978
|
+
}
|
|
979
|
+
let h13 = rayQueryGetCommittedIntersection(&rq13);
|
|
980
|
+
var c13 = vec3<f32>(0.0, 0.0, 50.0);
|
|
981
|
+
if (h13.kind != RAY_QUERY_INTERSECTION_NONE) {
|
|
982
|
+
let err = abs(h13.t - gdist);
|
|
983
|
+
if (err < gdist * 0.02 + 0.1) {
|
|
984
|
+
c13 = vec3<f32>(0.0, 50.0, 0.0);
|
|
985
|
+
} else {
|
|
986
|
+
c13 = vec3<f32>(50.0, 0.0, 0.0);
|
|
987
|
+
}
|
|
988
|
+
}
|
|
989
|
+
textureStore(out_hdr, px_full, vec4<f32>(c13, 1.0));
|
|
990
|
+
return;
|
|
991
|
+
}
|
|
992
|
+
if (debug == 14.0) {
|
|
993
|
+
// Shape probe: contour bands of the traced primary hit distance
|
|
994
|
+
// — shows what world the TLAS actually contains. Blue = miss.
|
|
995
|
+
let dir0 = normalize(p0 - u.cam_pos.xyz);
|
|
996
|
+
var rq14: ray_query;
|
|
997
|
+
rayQueryInitialize(&rq14, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
|
|
998
|
+
loop {
|
|
999
|
+
if (!rayQueryProceed(&rq14)) { break; }
|
|
1000
|
+
}
|
|
1001
|
+
let h14 = rayQueryGetCommittedIntersection(&rq14);
|
|
1002
|
+
var c14 = vec3<f32>(0.0, 0.0, 30.0);
|
|
1003
|
+
if (h14.kind != RAY_QUERY_INTERSECTION_NONE) {
|
|
1004
|
+
c14 = vec3<f32>(
|
|
1005
|
+
fract(h14.t * 0.125),
|
|
1006
|
+
fract(h14.t * 0.03125),
|
|
1007
|
+
fract(h14.t * 0.0078125),
|
|
1008
|
+
) * 20.0;
|
|
1009
|
+
}
|
|
1010
|
+
textureStore(out_hdr, px_full, vec4<f32>(c14, 1.0));
|
|
1011
|
+
return;
|
|
1012
|
+
}
|
|
1013
|
+
if (debug == 15.0) {
|
|
1014
|
+
// Aliasing probe: two queries, two very different rays.
|
|
1015
|
+
// A = primary (per-pixel), B = straight down (t ~= camera
|
|
1016
|
+
// height, near-constant). R channel = banded tA, G = banded tB.
|
|
1017
|
+
// If R == G everywhere the two queries alias to one object.
|
|
1018
|
+
let dirA = normalize(p0 - u.cam_pos.xyz);
|
|
1019
|
+
var rqA: ray_query;
|
|
1020
|
+
rayQueryInitialize(&rqA, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dirA));
|
|
1021
|
+
loop {
|
|
1022
|
+
if (!rayQueryProceed(&rqA)) { break; }
|
|
1023
|
+
}
|
|
1024
|
+
var rqB: ray_query;
|
|
1025
|
+
rayQueryInitialize(&rqB, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, vec3<f32>(0.0, -1.0, 0.0)));
|
|
1026
|
+
loop {
|
|
1027
|
+
if (!rayQueryProceed(&rqB)) { break; }
|
|
1028
|
+
}
|
|
1029
|
+
let hA = rayQueryGetCommittedIntersection(&rqA);
|
|
1030
|
+
let hB = rayQueryGetCommittedIntersection(&rqB);
|
|
1031
|
+
var tA = -1.0;
|
|
1032
|
+
var tB = -1.0;
|
|
1033
|
+
if (hA.kind != RAY_QUERY_INTERSECTION_NONE) { tA = hA.t; }
|
|
1034
|
+
if (hB.kind != RAY_QUERY_INTERSECTION_NONE) { tB = hB.t; }
|
|
1035
|
+
let c15 = vec3<f32>(fract(tA * 0.125) * 20.0, fract(tB * 0.125) * 20.0, 0.0);
|
|
1036
|
+
textureStore(out_hdr, px_full, vec4<f32>(c15, 1.0));
|
|
1037
|
+
return;
|
|
1038
|
+
}
|
|
1039
|
+
if (debug == 16.0) {
|
|
1040
|
+
// Raw numeric dump: traced primary intersection into the accum
|
|
1041
|
+
// buffer as (t, instance_custom_data, primitive_index, kind).
|
|
1042
|
+
// The CPU side reads a window of this buffer back and writes a
|
|
1043
|
+
// text file — no tonemap guesswork.
|
|
1044
|
+
let dir0 = normalize(p0 - u.cam_pos.xyz);
|
|
1045
|
+
var rq16: ray_query;
|
|
1046
|
+
rayQueryInitialize(&rq16, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
|
|
1047
|
+
loop {
|
|
1048
|
+
if (!rayQueryProceed(&rq16)) { break; }
|
|
1049
|
+
}
|
|
1050
|
+
let h16 = rayQueryGetCommittedIntersection(&rq16);
|
|
1051
|
+
let idx16 = gid.y * u.size.x + gid.x;
|
|
1052
|
+
accum_out[idx16] = vec4<f32>(
|
|
1053
|
+
h16.t,
|
|
1054
|
+
f32(h16.instance_custom_data),
|
|
1055
|
+
f32(h16.primitive_index),
|
|
1056
|
+
f32(h16.kind),
|
|
1057
|
+
);
|
|
1058
|
+
textureStore(out_hdr, px_full, vec4<f32>(0.2, 0.0, 0.4, 1.0));
|
|
1059
|
+
return;
|
|
1060
|
+
}
|
|
1061
|
+
if (debug == 17.0) {
|
|
1062
|
+
// Raw ray-generation dump: reconstructed world position + raw
|
|
1063
|
+
// depth, straight into accum for CPU readback.
|
|
1064
|
+
let idx17 = gid.y * u.size.x + gid.x;
|
|
1065
|
+
accum_out[idx17] = vec4<f32>(p0, depth);
|
|
1066
|
+
textureStore(out_hdr, px_full, vec4<f32>(0.4, 0.2, 0.0, 1.0));
|
|
1067
|
+
return;
|
|
1068
|
+
}
|
|
1069
|
+
if (debug == 18.0) {
|
|
1070
|
+
// Hybrid-sun validation: cascade shadow visibility at the
|
|
1071
|
+
// primary surface. Must match the raster shadow shapes exactly
|
|
1072
|
+
// (crisp tree/building shadows). Gray everywhere = the VPs are
|
|
1073
|
+
// wrong (transposition or stale).
|
|
1074
|
+
let vis18 = sun_vis_cascade(p0 + n0 * 0.02);
|
|
1075
|
+
textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(vis18 * 50.0), 1.0));
|
|
1076
|
+
return;
|
|
1077
|
+
}
|
|
1078
|
+
if (debug == 19.0) {
|
|
1079
|
+
// Numeric dump: cascade-0 shadow projection (ndc.xyz) + the
|
|
1080
|
+
// stored atlas depth at the landing texel (-1 = out of range,
|
|
1081
|
+
// -9 = degenerate w). Read back via pt_trace_dump.txt.
|
|
1082
|
+
let pw19 = p0 + n0 * 0.02;
|
|
1083
|
+
let clip19 = u.shadow_vps[0] * vec4<f32>(pw19, 1.0);
|
|
1084
|
+
var out19 = vec4<f32>(-9.0);
|
|
1085
|
+
if (abs(clip19.w) > 1e-6) {
|
|
1086
|
+
let ndc19 = clip19.xyz / clip19.w;
|
|
1087
|
+
let uv19 = vec2<f32>(ndc19.x * 0.5 + 0.5, 0.5 - ndc19.y * 0.5);
|
|
1088
|
+
var stored19 = -1.0;
|
|
1089
|
+
if (uv19.x >= 0.0 && uv19.x <= 1.0 && uv19.y >= 0.0 && uv19.y <= 1.0) {
|
|
1090
|
+
let dims19 = vec2<f32>(textureDimensions(shadow_atlas_0));
|
|
1091
|
+
stored19 = textureLoad(shadow_atlas_0, vec2<i32>(uv19 * dims19), 0);
|
|
1092
|
+
}
|
|
1093
|
+
out19 = vec4<f32>(ndc19.xyz, stored19);
|
|
1094
|
+
}
|
|
1095
|
+
accum_out[gid.y * u.size.x + gid.x] = out19;
|
|
1096
|
+
textureStore(out_hdr, px_full, vec4<f32>(0.1, 0.0, 0.2, 1.0));
|
|
1097
|
+
return;
|
|
1098
|
+
}
|
|
1099
|
+
if (debug == 6.0 || debug == 7.0) {
|
|
1100
|
+
// PT-2 validation: trace the primary ray through the TLAS
|
|
1101
|
+
// (ignoring the G-buffer) and show interpolated attributes.
|
|
1102
|
+
// Should match debug 2/3 up to smooth-vs-screen normals and
|
|
1103
|
+
// card-vs-texture resolution.
|
|
1104
|
+
let dir0 = normalize(p0 - u.cam_pos.xyz);
|
|
1105
|
+
var rq0: ray_query;
|
|
1106
|
+
rayQueryInitialize(&rq0, accel, RayDesc(0u, 0xFFu, 0.001, 1000.0, u.cam_pos.xyz, dir0));
|
|
1107
|
+
loop {
|
|
1108
|
+
if (!rayQueryProceed(&rq0)) { break; }
|
|
1109
|
+
}
|
|
1110
|
+
let h = rayQueryGetCommittedIntersection(&rq0);
|
|
1111
|
+
var col = vec3<f32>(1.0, 0.0, 1.0); // magenta: TLAS miss
|
|
1112
|
+
if (h.kind != RAY_QUERY_INTERSECTION_NONE) {
|
|
1113
|
+
let hinst = instance_data[h.instance_custom_data];
|
|
1114
|
+
if (hinst.geo.z > 0u) {
|
|
1115
|
+
let attrs = fetch_hit_attrs(hinst.geo, h.primitive_index, h.barycentrics);
|
|
1116
|
+
if (debug == 6.0) {
|
|
1117
|
+
col = normal_to_world(attrs.normal_os, h.world_to_object) * 0.5 + vec3<f32>(0.5);
|
|
1118
|
+
} else if (PT_HAS_TEXTURES) {
|
|
1119
|
+
col = hinst.albedo * pt_tex_sample(hinst.geo.w, attrs.uv);
|
|
1120
|
+
} else {
|
|
1121
|
+
col = vec3<f32>(1.0, 1.0, 0.0); // yellow: no tex arrays
|
|
1122
|
+
}
|
|
1123
|
+
} else {
|
|
1124
|
+
col = vec3<f32>(1.0, 0.5, 0.0); // orange: no geo window
|
|
1125
|
+
}
|
|
1126
|
+
}
|
|
1127
|
+
textureStore(out_hdr, px_full, vec4<f32>(col, 1.0));
|
|
1128
|
+
return;
|
|
1129
|
+
}
|
|
1130
|
+
|
|
1131
|
+
// ---- one path sample --------------------------------------------------
|
|
1132
|
+
|
|
1133
|
+
// Primary surface material from the G-buffer (R = metallic,
|
|
1134
|
+
// G = roughness). NEE stays diffuse-only, so scale it by
|
|
1135
|
+
// (1 - metallic) — metals have no diffuse lobe. Specular NEE is a
|
|
1136
|
+
// known gap (see the PT-2 ticket); specular reflection of sky and
|
|
1137
|
+
// scene comes from the GGX bounce below.
|
|
1138
|
+
let mr0 = textureLoad(material_tex, px_full, 0).rg;
|
|
1139
|
+
var metal_cur = mr0.r;
|
|
1140
|
+
var rough_cur = mr0.g;
|
|
1141
|
+
// Realtime mode samples the primary sun cone with structured IGN
|
|
1142
|
+
// noise, rolling per frame: spatially well-distributed (a 5x5
|
|
1143
|
+
// filter averages it nearly flat, unlike white PCG noise) and
|
|
1144
|
+
// temporally unbiased so the SVGF accumulation converges to the
|
|
1145
|
+
// true mean. Under the hybrid cascade sun this path only matters
|
|
1146
|
+
// when shadow maps are disabled. Progressive keeps white noise.
|
|
1147
|
+
var sun_r2 = rand_2f();
|
|
1148
|
+
if (u.cfg.x >= 2.0) {
|
|
1149
|
+
sun_r2 = vec2<f32>(
|
|
1150
|
+
ign_at(px_full, u.size.z),
|
|
1151
|
+
ign_at(px_full + vec2<i32>(17, 59), u.size.z),
|
|
1152
|
+
);
|
|
1153
|
+
}
|
|
1154
|
+
var view_cur = normalize(u.cam_pos.xyz - p0);
|
|
1155
|
+
// Reproject this surface into the previous trace grid ONCE — the
|
|
1156
|
+
// ReSTIR temporal reuse and the SVGF colour accumulation below both
|
|
1157
|
+
// consume the result (rp_* privates).
|
|
1158
|
+
compute_reproj(p0, px_full, depth);
|
|
1159
|
+
// PT-4 experimental: ext.w routes the primary point-light NEE
|
|
1160
|
+
// through the ReSTIR reservoirs; sun NEE is untouched either way.
|
|
1161
|
+
let use_restir = u.ext.w == 1u && u.cfg.x >= 2.0;
|
|
1162
|
+
// Sun + point lights, diffuse AND specular (nee_spec inside) — the
|
|
1163
|
+
// GGX highlight rides the same visibility as the diffuse term.
|
|
1164
|
+
var radiance = direct_light(
|
|
1165
|
+
p0 + n0 * 0.02, n0, albedo0 * (1.0 - metal_cur), sun_r2,
|
|
1166
|
+
view_cur, albedo0, rough_cur, metal_cur, !use_restir,
|
|
1167
|
+
);
|
|
1168
|
+
if (use_restir) {
|
|
1169
|
+
radiance += restir_point_light(
|
|
1170
|
+
gid.y * u.size.x + gid.x, p0 + n0 * 0.02, n0, view_cur,
|
|
1171
|
+
albedo0 * (1.0 - metal_cur), albedo0, rough_cur, metal_cur,
|
|
1172
|
+
);
|
|
1173
|
+
}
|
|
1174
|
+
var throughput = vec3<f32>(1.0);
|
|
1175
|
+
var origin = p0 + n0 * 0.02;
|
|
1176
|
+
var n_cur = n0;
|
|
1177
|
+
var alb_cur = albedo0;
|
|
1178
|
+
|
|
1179
|
+
let max_bounces = u32(u.cfg.y);
|
|
1180
|
+
for (var b = 0u; b < max_bounces; b = b + 1u) {
|
|
1181
|
+
let s = sample_brdf(n_cur, view_cur, alb_cur, rough_cur, metal_cur);
|
|
1182
|
+
if (!s.valid) {
|
|
1183
|
+
break;
|
|
1184
|
+
}
|
|
1185
|
+
throughput *= s.weight;
|
|
1186
|
+
let dir = s.dir;
|
|
1187
|
+
|
|
1188
|
+
var rq: ray_query;
|
|
1189
|
+
rayQueryInitialize(&rq, accel, RayDesc(0u, 0xFFu, 0.001, 500.0, origin, dir));
|
|
1190
|
+
loop {
|
|
1191
|
+
if (!rayQueryProceed(&rq)) { break; }
|
|
1192
|
+
}
|
|
1193
|
+
let hit = rayQueryGetCommittedIntersection(&rq);
|
|
1194
|
+
|
|
1195
|
+
if (hit.kind == RAY_QUERY_INTERSECTION_NONE) {
|
|
1196
|
+
radiance += throughput * sky_radiance(dir);
|
|
1197
|
+
break;
|
|
1198
|
+
}
|
|
1199
|
+
|
|
1200
|
+
let inst = instance_data[hit.instance_custom_data];
|
|
1201
|
+
let hit_ws = origin + dir * hit.t;
|
|
1202
|
+
let hit_os = (hit.world_to_object * vec4<f32>(hit_ws, 1.0)).xyz;
|
|
1203
|
+
|
|
1204
|
+
// PT-2: interpolated vertex normal + textured albedo when the
|
|
1205
|
+
// instance carries a geometry window; PT-1 flat-normal/card
|
|
1206
|
+
// fallback otherwise.
|
|
1207
|
+
var n_hit: vec3<f32>;
|
|
1208
|
+
var alb_hit: vec3<f32>;
|
|
1209
|
+
if (inst.geo.z > 0u) {
|
|
1210
|
+
let attrs = fetch_hit_attrs(inst.geo, hit.primitive_index, hit.barycentrics);
|
|
1211
|
+
n_hit = normal_to_world(attrs.normal_os, hit.world_to_object);
|
|
1212
|
+
if (PT_HAS_TEXTURES) {
|
|
1213
|
+
alb_hit = inst.albedo * pt_tex_sample(inst.geo.w, attrs.uv);
|
|
1214
|
+
} else {
|
|
1215
|
+
alb_hit = albedo_at_hit(inst, hit_os, dir);
|
|
1216
|
+
}
|
|
1217
|
+
} else {
|
|
1218
|
+
var nf = inst.normal_ws;
|
|
1219
|
+
let n_len = length(nf);
|
|
1220
|
+
if (n_len < 1e-4) { nf = -dir; } else { nf = nf / n_len; }
|
|
1221
|
+
n_hit = nf;
|
|
1222
|
+
alb_hit = albedo_at_hit(inst, hit_os, dir);
|
|
1223
|
+
}
|
|
1224
|
+
// A backface (or a flat normal pointing away) still bounces
|
|
1225
|
+
// outward, matching the OPAQUE two-sided raster convention.
|
|
1226
|
+
if (dot(n_hit, dir) > 0.0) { n_hit = -n_hit; }
|
|
1227
|
+
|
|
1228
|
+
// Emissive surfaces radiate; matches the Lumen fallback semantics
|
|
1229
|
+
// (albedo * emissive_luma).
|
|
1230
|
+
radiance += throughput * inst.albedo * inst.emissive_luma;
|
|
1231
|
+
|
|
1232
|
+
let hit_p = hit_ws + n_hit * 0.02;
|
|
1233
|
+
// view at a bounce vertex = back along the incoming ray.
|
|
1234
|
+
// Bounce vertices always use plain NEE (reservoirs are per
|
|
1235
|
+
// PRIMARY texel; reusing them off-surface would be biased).
|
|
1236
|
+
radiance += throughput * direct_light(
|
|
1237
|
+
hit_p, n_hit, alb_hit * (1.0 - inst.mat_params.y), rand_2f(),
|
|
1238
|
+
-dir, alb_hit, inst.mat_params.x, inst.mat_params.y, true,
|
|
1239
|
+
);
|
|
1240
|
+
|
|
1241
|
+
origin = hit_p;
|
|
1242
|
+
n_cur = n_hit;
|
|
1243
|
+
alb_cur = alb_hit;
|
|
1244
|
+
rough_cur = inst.mat_params.x;
|
|
1245
|
+
metal_cur = inst.mat_params.y;
|
|
1246
|
+
view_cur = -dir;
|
|
1247
|
+
|
|
1248
|
+
// Russian roulette from the third bounce.
|
|
1249
|
+
if (b >= 2u) {
|
|
1250
|
+
let q = clamp(max(throughput.r, max(throughput.g, throughput.b)), 0.05, 0.95);
|
|
1251
|
+
if (rand_f() > q) { break; }
|
|
1252
|
+
throughput /= q;
|
|
1253
|
+
}
|
|
1254
|
+
}
|
|
1255
|
+
|
|
1256
|
+
// NaN/Inf guard so one bad sample cannot poison the accumulator.
|
|
1257
|
+
if (radiance.r != radiance.r || radiance.g != radiance.g || radiance.b != radiance.b) {
|
|
1258
|
+
radiance = vec3<f32>(0.0);
|
|
1259
|
+
}
|
|
1260
|
+
|
|
1261
|
+
// ---- accumulate ---------------------------------------------------------
|
|
1262
|
+
|
|
1263
|
+
let idx = gid.y * u.size.x + gid.x;
|
|
1264
|
+
let mode = u.cfg.x;
|
|
1265
|
+
var prev = accum[idx];
|
|
1266
|
+
if (u.size.w == 0u) { prev = vec4<f32>(0.0); }
|
|
1267
|
+
|
|
1268
|
+
var out: vec3<f32>;
|
|
1269
|
+
if (mode >= 2.0) {
|
|
1270
|
+
// SVGF temporal accumulation (Schied et al. 2017). History and
|
|
1271
|
+
// output store IRRADIANCE (radiance demodulated by the primary
|
|
1272
|
+
// albedo) so the wavelet passes filter lighting only; the
|
|
1273
|
+
// final pass re-multiplies by the full-res G-buffer albedo.
|
|
1274
|
+
var irr = radiance / max(albedo0, vec3<f32>(0.05));
|
|
1275
|
+
// Firefly clamp — the one practical deviation from the paper
|
|
1276
|
+
// (reference implementations keep one too): it must bind in
|
|
1277
|
+
// IRRADIANCE space, because dividing by a dark albedo
|
|
1278
|
+
// amplifies radiance outliers up to 20x. Sunlit irradiance
|
|
1279
|
+
// sits around 1-3; 4 leaves real highlights alone.
|
|
1280
|
+
let irr_luma = dot(irr, vec3<f32>(0.2126, 0.7152, 0.0722));
|
|
1281
|
+
if (irr_luma > 4.0) {
|
|
1282
|
+
irr *= 4.0 / irr_luma;
|
|
1283
|
+
}
|
|
1284
|
+
let l_new = min(irr_luma, 4.0);
|
|
1285
|
+
|
|
1286
|
+
// Reprojection: 2x2 BILINEAR taps around the reprojected
|
|
1287
|
+
// position, each tap validated for geometric consistency
|
|
1288
|
+
// (relative linearized depth against the moments buffer).
|
|
1289
|
+
// Point sampling here quantizes to whole trace texels and
|
|
1290
|
+
// forced the old loose-tolerance workaround; weighted taps
|
|
1291
|
+
// give sub-texel reprojection and a honest per-tap test.
|
|
1292
|
+
var hist_rgb = vec3<f32>(0.0);
|
|
1293
|
+
var hist_m1 = 0.0;
|
|
1294
|
+
var hist_m2 = 0.0;
|
|
1295
|
+
var hist_n = 0.0;
|
|
1296
|
+
var wsum = 0.0;
|
|
1297
|
+
// Footprint depth window (current frame): one trace texel
|
|
1298
|
+
// covers ~3x3 full-res pixels; used below to tell a jitter
|
|
1299
|
+
// surface-flip apart from a true disocclusion.
|
|
1300
|
+
var fp_lo = 1e30;
|
|
1301
|
+
var fp_hi = 0.0;
|
|
1302
|
+
{
|
|
1303
|
+
let rx = max(i32(u.ext.x) / i32(u.size.x), 1);
|
|
1304
|
+
let ry = max(i32(u.ext.y) / i32(u.size.y), 1);
|
|
1305
|
+
for (var sy = 0; sy <= 1; sy = sy + 1) {
|
|
1306
|
+
for (var sx = 0; sx <= 1; sx = sx + 1) {
|
|
1307
|
+
let sp = min(
|
|
1308
|
+
px_full + vec2<i32>(sx * (rx - 1), sy * (ry - 1)),
|
|
1309
|
+
vec2<i32>(i32(u.ext.x) - 1, i32(u.ext.y) - 1),
|
|
1310
|
+
);
|
|
1311
|
+
let dz = depth_at(sp);
|
|
1312
|
+
if (dz >= 0.9999999) { continue; }
|
|
1313
|
+
let zl = lin_depth(dz);
|
|
1314
|
+
fp_lo = min(fp_lo, zl);
|
|
1315
|
+
fp_hi = max(fp_hi, zl);
|
|
1316
|
+
}
|
|
1317
|
+
}
|
|
1318
|
+
}
|
|
1319
|
+
// Reprojection basis was computed once at the top of the frame
|
|
1320
|
+
// (rp_* privates, shared with the ReSTIR temporal reuse).
|
|
1321
|
+
if (rp_valid && debug == 23.0) {
|
|
1322
|
+
// Reprojection dump: where this texel thinks it was
|
|
1323
|
+
// last frame (trace-grid units) + the depth pair the
|
|
1324
|
+
// acceptance test compares. Static camera => pos
|
|
1325
|
+
// must equal the texel's own coordinates.
|
|
1326
|
+
accum_out[idx] = vec4<f32>(
|
|
1327
|
+
f32(rp_base.x) + rp_fr.x, f32(rp_base.y) + rp_fr.y,
|
|
1328
|
+
rp_zl_here, lin_depth(moments[rp_nearest].w));
|
|
1329
|
+
moments_out[idx] = vec4<f32>(0.0, 0.0, 0.0, depth);
|
|
1330
|
+
return;
|
|
1331
|
+
}
|
|
1332
|
+
if (rp_valid) {
|
|
1333
|
+
// Tap test is TIGHT (surface identity). Cross-surface
|
|
1334
|
+
// blending is the expensive error: it leaks bright
|
|
1335
|
+
// blade-top lighting onto the ground below (gray-blue
|
|
1336
|
+
// mottle). Surface flips are handled after the loop,
|
|
1337
|
+
// not by widening this tolerance.
|
|
1338
|
+
let tol = 0.1 * rp_zl_here + 0.02;
|
|
1339
|
+
for (var ty = 0; ty <= 1; ty = ty + 1) {
|
|
1340
|
+
for (var tx = 0; tx <= 1; tx = tx + 1) {
|
|
1341
|
+
let q = rp_base + vec2<i32>(tx, ty);
|
|
1342
|
+
if (q.x < 0 || q.y < 0 || q.x >= i32(u.size.x) || q.y >= i32(u.size.y)) {
|
|
1343
|
+
continue;
|
|
1344
|
+
}
|
|
1345
|
+
let qidx = u32(q.y) * u.size.x + u32(q.x);
|
|
1346
|
+
let m = moments[qidx];
|
|
1347
|
+
// Sky texels and depth-inconsistent taps carry
|
|
1348
|
+
// another surface's lighting — skip them.
|
|
1349
|
+
if (m.w >= 0.9999999) {
|
|
1350
|
+
continue;
|
|
1351
|
+
}
|
|
1352
|
+
let zl_hist = lin_depth(m.w);
|
|
1353
|
+
if (abs(zl_hist - rp_zl_here) > tol) {
|
|
1354
|
+
continue;
|
|
1355
|
+
}
|
|
1356
|
+
let wx = mix(1.0 - rp_fr.x, rp_fr.x, f32(tx));
|
|
1357
|
+
let wy = mix(1.0 - rp_fr.y, rp_fr.y, f32(ty));
|
|
1358
|
+
let wt = wx * wy + 1e-4;
|
|
1359
|
+
hist_rgb += accum[qidx].rgb * wt;
|
|
1360
|
+
hist_m1 += m.x * wt;
|
|
1361
|
+
hist_m2 += m.y * wt;
|
|
1362
|
+
hist_n += m.z * wt;
|
|
1363
|
+
wsum += wt;
|
|
1364
|
+
}
|
|
1365
|
+
}
|
|
1366
|
+
}
|
|
1367
|
+
|
|
1368
|
+
// No tap matched. One trace texel holds ONE surface's history;
|
|
1369
|
+
// TAA jitter re-picks which surface the owner pixel sees each
|
|
1370
|
+
// frame on sub-texel geometry (grass). If the STORED surface
|
|
1371
|
+
// still exists in this texel's current footprint, this frame's
|
|
1372
|
+
// sample simply belongs to the other surface: keep the history
|
|
1373
|
+
// verbatim and drop the sample (blending would leak lighting
|
|
1374
|
+
// across surfaces; resetting would pin the texel at 1 spp and
|
|
1375
|
+
// fill the screen with speckle). The upsampler routes trace
|
|
1376
|
+
// texels to full-res pixels by depth, so the other surface
|
|
1377
|
+
// draws its lighting from neighbouring texels. Only a stored
|
|
1378
|
+
// surface that has LEFT the footprint is a true disocclusion.
|
|
1379
|
+
if (wsum <= 1e-3 && rp_valid) {
|
|
1380
|
+
let mnp = moments[rp_nearest];
|
|
1381
|
+
if (mnp.w < 0.9999999 && mnp.z > 0.0 && fp_hi > 0.0 && fp_lo < 1e29) {
|
|
1382
|
+
let zl_st = lin_depth(mnp.w);
|
|
1383
|
+
let wtol = 0.1 * zl_st + 0.02;
|
|
1384
|
+
if (zl_st > fp_lo - wtol && zl_st < fp_hi + wtol) {
|
|
1385
|
+
accum_out[idx] = accum[rp_nearest];
|
|
1386
|
+
moments_out[idx] = mnp;
|
|
1387
|
+
if (debug == 22.0) {
|
|
1388
|
+
// Numeric dump: n / variance / 99 = flip path / depth.
|
|
1389
|
+
accum_out[idx] = vec4<f32>(mnp.z, max(mnp.y - mnp.x * mnp.x, 0.0), 99.0, mnp.w);
|
|
1390
|
+
}
|
|
1391
|
+
if (debug == 20.0) {
|
|
1392
|
+
textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(mnp.z / 32.0), 1.0));
|
|
1393
|
+
}
|
|
1394
|
+
if (debug == 21.0) {
|
|
1395
|
+
let v_st = max(mnp.y - mnp.x * mnp.x, 0.0);
|
|
1396
|
+
textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(v_st * 10.0), 1.0));
|
|
1397
|
+
}
|
|
1398
|
+
return;
|
|
1399
|
+
}
|
|
1400
|
+
}
|
|
1401
|
+
}
|
|
1402
|
+
|
|
1403
|
+
var n_hist = 0.0;
|
|
1404
|
+
var seeded = false;
|
|
1405
|
+
if (wsum > 1e-3) {
|
|
1406
|
+
hist_rgb /= wsum;
|
|
1407
|
+
hist_m1 /= wsum;
|
|
1408
|
+
hist_m2 /= wsum;
|
|
1409
|
+
n_hist = hist_n / wsum;
|
|
1410
|
+
} else {
|
|
1411
|
+
// PT-9 — disocclusion seeding. A true disocclusion used to start
|
|
1412
|
+
// from the raw 1-spp sample with variance EXACTLY zero (m2 - m1²
|
|
1413
|
+
// of a single value), so freshly streamed-in pixels — the whole
|
|
1414
|
+
// viewport periphery whenever the camera translates — showed raw
|
|
1415
|
+
// path noise until history rebuilt. Indoors (bounce-dominated
|
|
1416
|
+
// light) that read as a salt-and-pepper ring around a clean
|
|
1417
|
+
// centre. But a newborn's NEIGHBOURS usually hold converged
|
|
1418
|
+
// history for the very same surface (the wall streaming in at
|
|
1419
|
+
// the screen edge continues inward), and accum[]/moments[] are
|
|
1420
|
+
// the previous frame's buffers — race-free to read at any texel.
|
|
1421
|
+
// Borrow from depth-consistent, converged (n ≥ 4) neighbours,
|
|
1422
|
+
// weighted by their history length; inherit HALF their history
|
|
1423
|
+
// (capped at 8) so the canonical 1/N blend still folds real
|
|
1424
|
+
// fresh samples in quickly. A pixel with no depth-compatible
|
|
1425
|
+
// neighbour (a genuinely new surface, e.g. rounding a doorway)
|
|
1426
|
+
// keeps the honest raw start — there is nothing to borrow.
|
|
1427
|
+
let zl_here = lin_depth(depth);
|
|
1428
|
+
let btol = 0.1 * zl_here + 0.02;
|
|
1429
|
+
var seed_rgb = vec3<f32>(0.0);
|
|
1430
|
+
var seed_m1 = 0.0;
|
|
1431
|
+
var seed_m2 = 0.0;
|
|
1432
|
+
var seed_n = 0.0;
|
|
1433
|
+
var seed_w = 0.0;
|
|
1434
|
+
// 7x7: under TRANSLATION a whole COLUMN of texels streams in per
|
|
1435
|
+
// frame (rotation only trickles a few px), so a 5x5 often found
|
|
1436
|
+
// nothing but fellow newborns and the band stayed raw.
|
|
1437
|
+
for (var by = -3; by <= 3; by = by + 1) {
|
|
1438
|
+
for (var bx = -3; bx <= 3; bx = bx + 1) {
|
|
1439
|
+
if (bx == 0 && by == 0) { continue; }
|
|
1440
|
+
let q = vec2<i32>(i32(gid.x) + bx, i32(gid.y) + by);
|
|
1441
|
+
if (q.x < 0 || q.y < 0 || q.x >= i32(u.size.x) || q.y >= i32(u.size.y)) {
|
|
1442
|
+
continue;
|
|
1443
|
+
}
|
|
1444
|
+
let qidx = u32(q.y) * u.size.x + u32(q.x);
|
|
1445
|
+
let m = moments[qidx];
|
|
1446
|
+
if (m.w >= 0.9999999 || m.z < 4.0) { continue; }
|
|
1447
|
+
if (abs(lin_depth(m.w) - zl_here) > btol) { continue; }
|
|
1448
|
+
let wt = m.z;
|
|
1449
|
+
seed_rgb += accum[qidx].rgb * wt;
|
|
1450
|
+
seed_m1 += m.x * wt;
|
|
1451
|
+
seed_m2 += m.y * wt;
|
|
1452
|
+
seed_n += m.z * wt;
|
|
1453
|
+
seed_w += wt;
|
|
1454
|
+
}
|
|
1455
|
+
}
|
|
1456
|
+
if (seed_w > 0.0) {
|
|
1457
|
+
hist_rgb = seed_rgb / seed_w;
|
|
1458
|
+
hist_m1 = seed_m1 / seed_w;
|
|
1459
|
+
hist_m2 = seed_m2 / seed_w;
|
|
1460
|
+
n_hist = min((seed_n / seed_w) * 0.5, 8.0);
|
|
1461
|
+
seeded = true;
|
|
1462
|
+
}
|
|
1463
|
+
}
|
|
1464
|
+
// Canonical blend: cumulative average while the history is
|
|
1465
|
+
// young (alpha = 1/N), settling to EMA. The floor is 0.1
|
|
1466
|
+
// rather than the paper's 0.2: our trace is half-res with a
|
|
1467
|
+
// 2-bounce sky lottery as the dominant noise source, and the
|
|
1468
|
+
// deeper average halves the residual mottle. Direct sun comes
|
|
1469
|
+
// from the raster cascades (deterministic), so the slower EMA
|
|
1470
|
+
// only delays indirect/ambient changes (~10 frames).
|
|
1471
|
+
let n_new = min(n_hist + 1.0, 32.0);
|
|
1472
|
+
let alpha_c = max(1.0 / n_new, 0.1);
|
|
1473
|
+
let out_irr = mix(hist_rgb, irr, alpha_c);
|
|
1474
|
+
let m1 = mix(hist_m1, l_new, alpha_c);
|
|
1475
|
+
let m2 = mix(hist_m2, l_new * l_new, alpha_c);
|
|
1476
|
+
// Temporal luminance variance — the signal that drives the
|
|
1477
|
+
// wavelet filter's luminance sigma. Young history makes this
|
|
1478
|
+
// unreliable; the first à-trous iteration substitutes a
|
|
1479
|
+
// spatial estimate when n < 4 (accum.w carries n via moments).
|
|
1480
|
+
var variance = max(m2 - m1 * m1, 0.0);
|
|
1481
|
+
// A newborn with NOTHING to borrow is a 1-sample estimate whose true
|
|
1482
|
+
// variance is unknown — not zero, which is what m2 - m1² of a single
|
|
1483
|
+
// value degenerates to. Zero variance tells the wavelet the pixel is
|
|
1484
|
+
// CONVERGED, so the raw outlier survived every iteration: that was
|
|
1485
|
+
// the residual noise band on wide stream-ins under camera
|
|
1486
|
+
// translation. Write a frank variance instead so the à-trous blurs
|
|
1487
|
+
// these pixels hard; one converged frame later the real statistics
|
|
1488
|
+
// take over.
|
|
1489
|
+
if (n_new <= 1.5 && !seeded) {
|
|
1490
|
+
variance = max(l_new * l_new, 0.25);
|
|
1491
|
+
}
|
|
1492
|
+
accum_out[idx] = vec4<f32>(out_irr, variance);
|
|
1493
|
+
moments_out[idx] = vec4<f32>(m1, m2, n_new, depth);
|
|
1494
|
+
if (debug == 22.0) {
|
|
1495
|
+
// Numeric dump: n / variance / accepted tap mass / depth.
|
|
1496
|
+
accum_out[idx] = vec4<f32>(n_new, variance, wsum, depth);
|
|
1497
|
+
}
|
|
1498
|
+
if (debug == 20.0) {
|
|
1499
|
+
// History length heat: white = full 32-frame history.
|
|
1500
|
+
textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(n_new / 32.0), 1.0));
|
|
1501
|
+
}
|
|
1502
|
+
if (debug == 21.0) {
|
|
1503
|
+
// Variance view (x10 so typical values are visible).
|
|
1504
|
+
textureStore(out_hdr, px_full, vec4<f32>(vec3<f32>(variance * 10.0), 1.0));
|
|
1505
|
+
}
|
|
1506
|
+
return;
|
|
1507
|
+
} else {
|
|
1508
|
+
// Progressive keeps its firefly cap, relaxing with
|
|
1509
|
+
// accumulation depth (a deep average can absorb real energy).
|
|
1510
|
+
let luma = dot(radiance, vec3<f32>(0.2126, 0.7152, 0.0722));
|
|
1511
|
+
let cap = 4.0 + f32(min(u.size.w, 28u));
|
|
1512
|
+
if (luma > cap) { radiance *= cap / luma; }
|
|
1513
|
+
// Progressive: plain running sum; count lives on the CPU.
|
|
1514
|
+
// Ping-pong read/write at the same index (static camera only).
|
|
1515
|
+
let sum = prev.rgb + radiance;
|
|
1516
|
+
let n = f32(u.size.w) + 1.0;
|
|
1517
|
+
accum_out[idx] = vec4<f32>(sum, n);
|
|
1518
|
+
out = sum / n;
|
|
1519
|
+
// Interim gameplay behaviour until PT-3's denoiser: a moving
|
|
1520
|
+
// camera resets accumulation every frame, and raw 1-spp noise
|
|
1521
|
+
// through TSR looks terrible. Keep accumulating but leave the
|
|
1522
|
+
// raster frame on screen until a few samples exist — stand
|
|
1523
|
+
// still for half a second and PT dissolves in. The CPU side
|
|
1524
|
+
// mirrors this threshold (pt_wrote_frame) so SSGI/SSR stay on
|
|
1525
|
+
// for the raster frames.
|
|
1526
|
+
if (u.size.w < 8u) {
|
|
1527
|
+
return;
|
|
1528
|
+
}
|
|
1529
|
+
}
|
|
1530
|
+
|
|
1531
|
+
textureStore(out_hdr, px_full, vec4<f32>(out, 1.0));
|
|
1532
|
+
}
|
|
1533
|
+
"#;
|
|
1534
|
+
|
|
1535
|
+
/// PT-3b — SVGF wavelet filter (Schied et al. 2017) for the realtime
|
|
1536
|
+
/// mode. Four `cs_mid` à-trous iterations (steps 1/2/4/8) run on the
|
|
1537
|
+
/// trace grid over the temporally-accumulated irradiance; `cs_final`
|
|
1538
|
+
/// joint-bilaterally upsamples to full resolution and re-modulates the
|
|
1539
|
+
/// G-buffer albedo. Buffers: src/dst = (irradiance rgb, luminance
|
|
1540
|
+
/// variance w); `geo` = the kernel's moments buffer (mu1, mu2, history
|
|
1541
|
+
/// length, raw depth) — static across iterations, it carries the depth
|
|
1542
|
+
/// for edge-stopping and the sky marker (depth = far plane).
|
|
1543
|
+
///
|
|
1544
|
+
/// The luminance edge-stop is VARIANCE-DRIVEN: sigma_l scales with the
|
|
1545
|
+
/// per-texel noise estimate, so grainy regions blur hard while
|
|
1546
|
+
/// converged shading detail survives. Variance travels with the signal,
|
|
1547
|
+
/// filtered by the squared weights, shrinking each iteration exactly as
|
|
1548
|
+
/// the residual noise does. This replaces the old fixed sigma schedule,
|
|
1549
|
+
/// the despeckle clamp and the history spike clamp — with a correct
|
|
1550
|
+
/// variance estimate none of those are needed.
|
|
1551
|
+
pub(in crate::renderer) const PT_ATROUS_WGSL: &str = r#"
|
|
1552
|
+
struct AtrousParams {
|
|
1553
|
+
// x = step (texels), y = 1.0 on the FIRST iteration (enables the
|
|
1554
|
+
// short-history spatial variance fallback), z/w = trace dims
|
|
1555
|
+
p: vec4<f32>,
|
|
1556
|
+
// x/y = full G-buffer dims (cs_final upsamples trace -> full when
|
|
1557
|
+
// they differ), z/w unused.
|
|
1558
|
+
p2: vec4<f32>,
|
|
1559
|
+
};
|
|
1560
|
+
@group(0) @binding(0) var<uniform> ap: AtrousParams;
|
|
1561
|
+
@group(0) @binding(1) var<storage, read> src: array<vec4<f32>>;
|
|
1562
|
+
@group(0) @binding(2) var<storage, read_write> dst: array<vec4<f32>>;
|
|
1563
|
+
@group(0) @binding(3) var out_hdr_a: texture_storage_2d<rgba16float, write>;
|
|
1564
|
+
@group(0) @binding(4) var depth_full: texture_depth_2d;
|
|
1565
|
+
// Full-res G-buffer albedo: cs_final re-modulates the filtered
|
|
1566
|
+
// irradiance with it (SVGF demodulation keeps textures crisp).
|
|
1567
|
+
@group(0) @binding(5) var albedo_full: texture_2d<f32>;
|
|
1568
|
+
// Kernel moments buffer: (mu1, mu2, history length, raw depth).
|
|
1569
|
+
@group(0) @binding(6) var<storage, read> geo: array<vec4<f32>>;
|
|
1570
|
+
|
|
1571
|
+
fn lin_depth_a(d: f32) -> f32 {
|
|
1572
|
+
return 0.02 / max(1.0 - d, 1e-6);
|
|
1573
|
+
}
|
|
1574
|
+
|
|
1575
|
+
fn luma_of(c: vec3<f32>) -> f32 {
|
|
1576
|
+
return dot(c, vec3<f32>(0.2126, 0.7152, 0.0722));
|
|
1577
|
+
}
|
|
1578
|
+
|
|
1579
|
+
// B3-spline kernel weight for |offset| 0/1/2.
|
|
1580
|
+
fn kern(d: i32) -> f32 {
|
|
1581
|
+
let a = abs(d);
|
|
1582
|
+
if (a == 0) { return 0.375; }
|
|
1583
|
+
if (a == 1) { return 0.25; }
|
|
1584
|
+
return 0.0625;
|
|
1585
|
+
}
|
|
1586
|
+
|
|
1587
|
+
// SVGF luminance sigma (paper value).
|
|
1588
|
+
const SIGMA_L: f32 = 4.0;
|
|
1589
|
+
|
|
1590
|
+
fn filter_at(px: vec2<i32>, w: i32, h: i32, step: i32, first: bool) -> vec4<f32> {
|
|
1591
|
+
let cidx = u32(px.y) * u32(w) + u32(px.x);
|
|
1592
|
+
let g_c = geo[cidx];
|
|
1593
|
+
let center = src[cidx];
|
|
1594
|
+
if (g_c.w >= 0.9999999) {
|
|
1595
|
+
return center;
|
|
1596
|
+
}
|
|
1597
|
+
let zc = lin_depth_a(g_c.w);
|
|
1598
|
+
let lc = luma_of(center.rgb);
|
|
1599
|
+
|
|
1600
|
+
// Center variance. Temporal variance from a young history (< 4
|
|
1601
|
+
// frames, e.g. right after a disocclusion) is meaningless — the
|
|
1602
|
+
// paper substitutes a spatial luminance-variance estimate there.
|
|
1603
|
+
var var_c = max(center.w, 0.0);
|
|
1604
|
+
if (first && g_c.z < 4.0) {
|
|
1605
|
+
var s1 = 0.0;
|
|
1606
|
+
var s2 = 0.0;
|
|
1607
|
+
var cnt = 0.0;
|
|
1608
|
+
for (var dy = -1; dy <= 1; dy = dy + 1) {
|
|
1609
|
+
for (var dx = -1; dx <= 1; dx = dx + 1) {
|
|
1610
|
+
let q = px + vec2<i32>(dx, dy);
|
|
1611
|
+
if (q.x < 0 || q.y < 0 || q.x >= w || q.y >= h) {
|
|
1612
|
+
continue;
|
|
1613
|
+
}
|
|
1614
|
+
let qi = u32(q.y) * u32(w) + u32(q.x);
|
|
1615
|
+
if (geo[qi].w >= 0.9999999) {
|
|
1616
|
+
continue;
|
|
1617
|
+
}
|
|
1618
|
+
let lq = luma_of(src[qi].rgb);
|
|
1619
|
+
s1 += lq;
|
|
1620
|
+
s2 += lq * lq;
|
|
1621
|
+
cnt += 1.0;
|
|
1622
|
+
}
|
|
1623
|
+
}
|
|
1624
|
+
if (cnt > 1.0) {
|
|
1625
|
+
let mu = s1 / cnt;
|
|
1626
|
+
var_c = max(var_c, max(s2 / cnt - mu * mu, 0.0));
|
|
1627
|
+
}
|
|
1628
|
+
// Variance boost: a young history's estimate is unreliable in
|
|
1629
|
+
// BOTH directions, and underestimating is the expensive error
|
|
1630
|
+
// (a 1-spp outlier with variance 0 survives every iteration as
|
|
1631
|
+
// a false edge). Floor it so fresh texels blend spatially until
|
|
1632
|
+
// their temporal estimate matures.
|
|
1633
|
+
var_c = max(var_c, 0.25);
|
|
1634
|
+
}
|
|
1635
|
+
|
|
1636
|
+
// 3x3 gaussian prefilter of the variance (paper 4.2): keeps a
|
|
1637
|
+
// single hot texel from stopping its own smoothing.
|
|
1638
|
+
var vsum = var_c * 0.25;
|
|
1639
|
+
var vwsum = 0.25;
|
|
1640
|
+
for (var dy = -1; dy <= 1; dy = dy + 1) {
|
|
1641
|
+
for (var dx = -1; dx <= 1; dx = dx + 1) {
|
|
1642
|
+
if (dx == 0 && dy == 0) {
|
|
1643
|
+
continue;
|
|
1644
|
+
}
|
|
1645
|
+
let q = px + vec2<i32>(dx, dy);
|
|
1646
|
+
if (q.x < 0 || q.y < 0 || q.x >= w || q.y >= h) {
|
|
1647
|
+
continue;
|
|
1648
|
+
}
|
|
1649
|
+
let qi = u32(q.y) * u32(w) + u32(q.x);
|
|
1650
|
+
if (geo[qi].w >= 0.9999999) {
|
|
1651
|
+
continue;
|
|
1652
|
+
}
|
|
1653
|
+
let gw = select(0.0625, 0.125, dx == 0 || dy == 0);
|
|
1654
|
+
vsum += max(src[qi].w, 0.0) * gw;
|
|
1655
|
+
vwsum += gw;
|
|
1656
|
+
}
|
|
1657
|
+
}
|
|
1658
|
+
let sigma_l_denom = SIGMA_L * sqrt(max(vsum / vwsum, 0.0)) + 1e-3;
|
|
1659
|
+
|
|
1660
|
+
// Depth gradient (central differences on linear depth) scales the
|
|
1661
|
+
// depth edge-stop so steep-slope surfaces stay connected while
|
|
1662
|
+
// depth discontinuities still stop the filter (paper eq. 3).
|
|
1663
|
+
var dzdx = 0.0;
|
|
1664
|
+
var dzdy = 0.0;
|
|
1665
|
+
if (px.x > 0 && px.x < w - 1) {
|
|
1666
|
+
dzdx = (lin_depth_a(geo[cidx + 1u].w) - lin_depth_a(geo[cidx - 1u].w)) * 0.5;
|
|
1667
|
+
}
|
|
1668
|
+
if (px.y > 0 && px.y < h - 1) {
|
|
1669
|
+
dzdy = (lin_depth_a(geo[cidx + u32(w)].w) - lin_depth_a(geo[cidx - u32(w)].w)) * 0.5;
|
|
1670
|
+
}
|
|
1671
|
+
|
|
1672
|
+
var sum = vec3<f32>(0.0);
|
|
1673
|
+
var sum_v = 0.0;
|
|
1674
|
+
var wsum = 0.0;
|
|
1675
|
+
for (var dy = -2; dy <= 2; dy = dy + 1) {
|
|
1676
|
+
for (var dx = -2; dx <= 2; dx = dx + 1) {
|
|
1677
|
+
let q = px + vec2<i32>(dx, dy) * step;
|
|
1678
|
+
if (q.x < 0 || q.y < 0 || q.x >= w || q.y >= h) {
|
|
1679
|
+
continue;
|
|
1680
|
+
}
|
|
1681
|
+
let qi = u32(q.y) * u32(w) + u32(q.x);
|
|
1682
|
+
let g_q = geo[qi];
|
|
1683
|
+
if (g_q.w >= 0.9999999) {
|
|
1684
|
+
continue;
|
|
1685
|
+
}
|
|
1686
|
+
let s = src[qi];
|
|
1687
|
+
let zq = lin_depth_a(g_q.w);
|
|
1688
|
+
let z_denom = abs(dzdx) * f32(abs(dx) * step)
|
|
1689
|
+
+ abs(dzdy) * f32(abs(dy) * step)
|
|
1690
|
+
+ 0.01 * zc + 1e-4;
|
|
1691
|
+
let wz = exp(-abs(zq - zc) / z_denom);
|
|
1692
|
+
let wl = exp(-abs(luma_of(s.rgb) - lc) / sigma_l_denom);
|
|
1693
|
+
let wgt = kern(dx) * kern(dy) * wz * wl;
|
|
1694
|
+
sum += s.rgb * wgt;
|
|
1695
|
+
// Variance contracts with the SQUARED weights — the
|
|
1696
|
+
// estimate shrinks exactly as the filtered noise does.
|
|
1697
|
+
sum_v += max(s.w, 0.0) * wgt * wgt;
|
|
1698
|
+
wsum += wgt;
|
|
1699
|
+
}
|
|
1700
|
+
}
|
|
1701
|
+
if (wsum < 1e-6) {
|
|
1702
|
+
return center;
|
|
1703
|
+
}
|
|
1704
|
+
return vec4<f32>(sum / wsum, sum_v / (wsum * wsum));
|
|
1705
|
+
}
|
|
1706
|
+
|
|
1707
|
+
@compute @workgroup_size(8, 8, 1)
|
|
1708
|
+
fn cs_mid(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
1709
|
+
let w = i32(ap.p.z);
|
|
1710
|
+
let h = i32(ap.p.w);
|
|
1711
|
+
if (i32(gid.x) >= w || i32(gid.y) >= h) {
|
|
1712
|
+
return;
|
|
1713
|
+
}
|
|
1714
|
+
let px = vec2<i32>(i32(gid.x), i32(gid.y));
|
|
1715
|
+
dst[gid.y * u32(w) + gid.x] = filter_at(px, w, h, i32(ap.p.x), ap.p.y > 0.5);
|
|
1716
|
+
}
|
|
1717
|
+
|
|
1718
|
+
// Final pass runs at FULL resolution: depth-guided joint-bilateral
|
|
1719
|
+
// upsample of the filtered irradiance, re-modulated by the full-res
|
|
1720
|
+
// G-buffer albedo. Sky pixels (full-res depth at far plane) are never
|
|
1721
|
+
// written so the raster sky survives. When the trace grid IS the full
|
|
1722
|
+
// grid it degenerates to a plain modulate-and-write.
|
|
1723
|
+
@compute @workgroup_size(8, 8, 1)
|
|
1724
|
+
fn cs_final(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
1725
|
+
let fw = i32(ap.p2.x);
|
|
1726
|
+
let fh = i32(ap.p2.y);
|
|
1727
|
+
if (i32(gid.x) >= fw || i32(gid.y) >= fh) {
|
|
1728
|
+
return;
|
|
1729
|
+
}
|
|
1730
|
+
let px = vec2<i32>(i32(gid.x), i32(gid.y));
|
|
1731
|
+
let d = textureLoad(depth_full, px, 0);
|
|
1732
|
+
if (d >= 0.9999999) {
|
|
1733
|
+
return;
|
|
1734
|
+
}
|
|
1735
|
+
let hw = i32(ap.p.z);
|
|
1736
|
+
let hh = i32(ap.p.w);
|
|
1737
|
+
let alb = max(textureLoad(albedo_full, px, 0).rgb, vec3<f32>(0.05));
|
|
1738
|
+
if (hw == fw) {
|
|
1739
|
+
let cidx = gid.y * u32(fw) + gid.x;
|
|
1740
|
+
if (geo[cidx].w >= 0.9999999) {
|
|
1741
|
+
return;
|
|
1742
|
+
}
|
|
1743
|
+
textureStore(out_hdr_a, px, vec4<f32>(src[cidx].rgb * alb, 1.0));
|
|
1744
|
+
return;
|
|
1745
|
+
}
|
|
1746
|
+
// 3x3 taps around the trace texel, weighted by SUB-TEXEL tent distance
|
|
1747
|
+
// and relative linear-depth agreement with THIS full-res pixel. The
|
|
1748
|
+
// weights used to centre on the CONTAINING texel (integer mapping, no
|
|
1749
|
+
// fractional phase), which reconstructed the lighting as trace-texel-
|
|
1750
|
+
// sized constant blocks — and because the trace grid's sample phase
|
|
1751
|
+
// rotates every frame, the block boundaries CRAWLED under motion (the
|
|
1752
|
+
// residual "pixelated while moving" indoors). Tent weights at the
|
|
1753
|
+
// continuous position make this a true bilinear-plus-depth joint
|
|
1754
|
+
// bilateral: smooth gradients, stable under the phase rotation. The
|
|
1755
|
+
// epsilon keeps thin foreground geometry (whose taps all mismatch)
|
|
1756
|
+
// softly averaged rather than black.
|
|
1757
|
+
let zc = lin_depth_a(d);
|
|
1758
|
+
// Generalized ratio mapping (trace grid is budget-capped, not
|
|
1759
|
+
// always exactly half of full res), texel centres aligned.
|
|
1760
|
+
let fx = (f32(px.x) + 0.5) * f32(hw) / f32(fw) - 0.5;
|
|
1761
|
+
let fy = (f32(px.y) + 0.5) * f32(hh) / f32(fh) - 0.5;
|
|
1762
|
+
let bx = i32(floor(fx));
|
|
1763
|
+
let by = i32(floor(fy));
|
|
1764
|
+
let frx = fx - f32(bx);
|
|
1765
|
+
let fry = fy - f32(by);
|
|
1766
|
+
var sum = vec3<f32>(0.0);
|
|
1767
|
+
var wsum = 0.0;
|
|
1768
|
+
for (var dy = -1; dy <= 1; dy = dy + 1) {
|
|
1769
|
+
for (var dx = -1; dx <= 1; dx = dx + 1) {
|
|
1770
|
+
let qx = bx + dx;
|
|
1771
|
+
let qy = by + dy;
|
|
1772
|
+
if (qx < 0 || qy < 0 || qx >= hw || qy >= hh) {
|
|
1773
|
+
continue;
|
|
1774
|
+
}
|
|
1775
|
+
let qi = u32(qy) * u32(hw) + u32(qx);
|
|
1776
|
+
if (geo[qi].w >= 0.9999999) {
|
|
1777
|
+
continue;
|
|
1778
|
+
}
|
|
1779
|
+
let s = src[qi];
|
|
1780
|
+
let wx = max(0.0, 1.0 - abs(f32(dx) - frx));
|
|
1781
|
+
let wy = max(0.0, 1.0 - abs(f32(dy) - fry));
|
|
1782
|
+
let wz = exp(-abs(lin_depth_a(geo[qi].w) - zc) / (0.08 * zc + 0.02));
|
|
1783
|
+
let wgt = wx * wy * wz + 1e-5;
|
|
1784
|
+
sum += s.rgb * wgt;
|
|
1785
|
+
wsum += wgt;
|
|
1786
|
+
}
|
|
1787
|
+
}
|
|
1788
|
+
if (wsum < 1e-6) {
|
|
1789
|
+
return;
|
|
1790
|
+
}
|
|
1791
|
+
textureStore(out_hdr_a, px, vec4<f32>((sum / wsum) * alb, 1.0));
|
|
1792
|
+
}
|
|
1793
|
+
"#;
|
|
1794
|
+
|
|
1795
|
+
/// PT-6 — compute pre-skin: poses one skinned mesh into its PT geometry
|
|
1796
|
+
/// megabuffer window (world space; the joint palette bakes placement).
|
|
1797
|
+
/// The CPU wrote the bind-pose vertex data into the window this frame;
|
|
1798
|
+
/// this pass overwrites position + normal, leaving color/uv untouched.
|
|
1799
|
+
/// The BLAS for the dynamic instance then reads the same window
|
|
1800
|
+
/// (first_vertex offset), so intersection and hit shading share bytes.
|
|
1801
|
+
pub(in crate::renderer) const PT_SKIN_WGSL: &str = r#"
|
|
1802
|
+
struct SkinParams {
|
|
1803
|
+
// Places the rare rigid (weightless) verts, same as the raster VS.
|
|
1804
|
+
model: mat4x4<f32>,
|
|
1805
|
+
// x = megabuffer vertex slot base (Vertex3D units), y = vertex
|
|
1806
|
+
// count, z = joint palette base offset, w unused.
|
|
1807
|
+
p: vec4<u32>,
|
|
1808
|
+
};
|
|
1809
|
+
struct SkinJoints {
|
|
1810
|
+
m: array<mat4x4<f32>, 1024>,
|
|
1811
|
+
};
|
|
1812
|
+
@group(0) @binding(0) var<uniform> sp: SkinParams;
|
|
1813
|
+
@group(0) @binding(1) var<storage, read> src_v: array<f32>;
|
|
1814
|
+
@group(0) @binding(2) var<storage, read_write> dst_v: array<f32>;
|
|
1815
|
+
@group(0) @binding(3) var<uniform> joints: SkinJoints;
|
|
1816
|
+
|
|
1817
|
+
@compute @workgroup_size(64, 1, 1)
|
|
1818
|
+
fn cs_skin(@builtin(global_invocation_id) gid: vec3<u32>) {
|
|
1819
|
+
let i = gid.x;
|
|
1820
|
+
if (i >= sp.p.y) { return; }
|
|
1821
|
+
// Vertex3D words: pos +0, normal +3, color +6, uv +10, joints +12,
|
|
1822
|
+
// weights +16, tangent +20 (stride 24 f32).
|
|
1823
|
+
let s = i * 24u;
|
|
1824
|
+
let d = (sp.p.x + i) * 24u;
|
|
1825
|
+
let pos = vec4<f32>(src_v[s], src_v[s + 1u], src_v[s + 2u], 1.0);
|
|
1826
|
+
let nrm = vec4<f32>(src_v[s + 3u], src_v[s + 4u], src_v[s + 5u], 0.0);
|
|
1827
|
+
let w = vec4<f32>(
|
|
1828
|
+
src_v[s + 16u], src_v[s + 17u], src_v[s + 18u], src_v[s + 19u],
|
|
1829
|
+
);
|
|
1830
|
+
var wp: vec3<f32>;
|
|
1831
|
+
var wn: vec3<f32>;
|
|
1832
|
+
if (w.x + w.y + w.z + w.w > 0.01) {
|
|
1833
|
+
// Same palette blend as the raster VS (core.rs vs_main_scene).
|
|
1834
|
+
let j0 = u32(src_v[s + 12u]) + sp.p.z;
|
|
1835
|
+
let j1 = u32(src_v[s + 13u]) + sp.p.z;
|
|
1836
|
+
let j2 = u32(src_v[s + 14u]) + sp.p.z;
|
|
1837
|
+
let j3 = u32(src_v[s + 15u]) + sp.p.z;
|
|
1838
|
+
let m0 = joints.m[j0];
|
|
1839
|
+
let m1 = joints.m[j1];
|
|
1840
|
+
let m2 = joints.m[j2];
|
|
1841
|
+
let m3 = joints.m[j3];
|
|
1842
|
+
wp = ((m0 * pos) * w.x + (m1 * pos) * w.y
|
|
1843
|
+
+ (m2 * pos) * w.z + (m3 * pos) * w.w).xyz;
|
|
1844
|
+
wn = ((m0 * nrm) * w.x + (m1 * nrm) * w.y
|
|
1845
|
+
+ (m2 * nrm) * w.z + (m3 * nrm) * w.w).xyz;
|
|
1846
|
+
} else {
|
|
1847
|
+
wp = (sp.model * pos).xyz;
|
|
1848
|
+
wn = (sp.model * nrm).xyz;
|
|
1849
|
+
}
|
|
1850
|
+
let ln = length(wn);
|
|
1851
|
+
if (ln > 1e-6) { wn = wn / ln; }
|
|
1852
|
+
dst_v[d] = wp.x;
|
|
1853
|
+
dst_v[d + 1u] = wp.y;
|
|
1854
|
+
dst_v[d + 2u] = wp.z;
|
|
1855
|
+
dst_v[d + 3u] = wn.x;
|
|
1856
|
+
dst_v[d + 4u] = wn.y;
|
|
1857
|
+
dst_v[d + 5u] = wn.z;
|
|
1858
|
+
}
|
|
1859
|
+
"#;
|