@bornengine/engine 0.4.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (213) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +231 -0
  3. package/native/android/Cargo.lock +1848 -0
  4. package/native/android/Cargo.toml +24 -0
  5. package/native/android/src/lib.rs +702 -0
  6. package/native/ios/Cargo.lock +1690 -0
  7. package/native/ios/Cargo.toml +32 -0
  8. package/native/ios/src/lib.rs +1267 -0
  9. package/native/linux/Cargo.lock +3279 -0
  10. package/native/linux/Cargo.toml +29 -0
  11. package/native/linux/src/lib.rs +1331 -0
  12. package/native/macos/Cargo.lock +3310 -0
  13. package/native/macos/Cargo.toml +46 -0
  14. package/native/macos/src/lib.rs +1302 -0
  15. package/native/shared/Cargo.lock +1899 -0
  16. package/native/shared/Cargo.toml +62 -0
  17. package/native/shared/assets/default_font.ttf +0 -0
  18. package/native/shared/build.rs +270 -0
  19. package/native/shared/shaders/common/clouds.wgsl +122 -0
  20. package/native/shared/shaders/common/fog.wgsl +16 -0
  21. package/native/shared/shaders/common/foliage_wind.wgsl +98 -0
  22. package/native/shared/shaders/common/imposter.wgsl +112 -0
  23. package/native/shared/shaders/common/pbr.wgsl +186 -0
  24. package/native/shared/shaders/common/shadows.wgsl +186 -0
  25. package/native/shared/shaders/common/sky.wgsl +8 -0
  26. package/native/shared/shaders/common/tonemap.wgsl +25 -0
  27. package/native/shared/shaders/impulse_field.wgsl +57 -0
  28. package/native/shared/shaders/material_abi.wgsl +383 -0
  29. package/native/shared/shaders/materials/test_minimal.wgsl +42 -0
  30. package/native/shared/src/anim_mixer.rs +61 -0
  31. package/native/shared/src/attach.rs +263 -0
  32. package/native/shared/src/audio/decode.rs +123 -0
  33. package/native/shared/src/audio/mod.rs +863 -0
  34. package/native/shared/src/audio/render.rs +892 -0
  35. package/native/shared/src/audio/spsc.rs +156 -0
  36. package/native/shared/src/audio/stream.rs +226 -0
  37. package/native/shared/src/custom_shaders.rs +104 -0
  38. package/native/shared/src/decals.rs +245 -0
  39. package/native/shared/src/drs.rs +211 -0
  40. package/native/shared/src/engine.rs +261 -0
  41. package/native/shared/src/ffi.rs +116 -0
  42. package/native/shared/src/ffi_core/assets.rs +388 -0
  43. package/native/shared/src/ffi_core/audio_ffi.rs +184 -0
  44. package/native/shared/src/ffi_core/draw.rs +334 -0
  45. package/native/shared/src/ffi_core/game_loop.rs +577 -0
  46. package/native/shared/src/ffi_core/input.rs +234 -0
  47. package/native/shared/src/ffi_core/mod.rs +127 -0
  48. package/native/shared/src/ffi_core/models.rs +1154 -0
  49. package/native/shared/src/ffi_core/ragdoll_ffi.rs +261 -0
  50. package/native/shared/src/ffi_core/scene.rs +626 -0
  51. package/native/shared/src/ffi_core/vfx.rs +212 -0
  52. package/native/shared/src/ffi_core/visual.rs +691 -0
  53. package/native/shared/src/frame_callbacks.rs +122 -0
  54. package/native/shared/src/geometry.rs +236 -0
  55. package/native/shared/src/handles.rs +182 -0
  56. package/native/shared/src/input.rs +448 -0
  57. package/native/shared/src/jolt_sys.rs +822 -0
  58. package/native/shared/src/lib.rs +55 -0
  59. package/native/shared/src/models.rs +1093 -0
  60. package/native/shared/src/models_gltf.rs +1280 -0
  61. package/native/shared/src/particles.rs +391 -0
  62. package/native/shared/src/physics_jolt.rs +1908 -0
  63. package/native/shared/src/picking.rs +298 -0
  64. package/native/shared/src/postfx.rs +345 -0
  65. package/native/shared/src/profiler.rs +492 -0
  66. package/native/shared/src/ragdoll.rs +474 -0
  67. package/native/shared/src/renderer/atmosphere_lut.rs +573 -0
  68. package/native/shared/src/renderer/brdf_lut.rs +154 -0
  69. package/native/shared/src/renderer/draw2d.rs +143 -0
  70. package/native/shared/src/renderer/formats.rs +822 -0
  71. package/native/shared/src/renderer/froxel.rs +421 -0
  72. package/native/shared/src/renderer/gi_bake.rs +653 -0
  73. package/native/shared/src/renderer/graph.rs +462 -0
  74. package/native/shared/src/renderer/hiz.rs +269 -0
  75. package/native/shared/src/renderer/hot_reload.rs +390 -0
  76. package/native/shared/src/renderer/impulse_field.rs +456 -0
  77. package/native/shared/src/renderer/lighting.rs +154 -0
  78. package/native/shared/src/renderer/material_instancing.rs +171 -0
  79. package/native/shared/src/renderer/material_pipeline.rs +700 -0
  80. package/native/shared/src/renderer/material_system.rs +1996 -0
  81. package/native/shared/src/renderer/material_system_tests.rs +601 -0
  82. package/native/shared/src/renderer/material_system_wasm.rs +41 -0
  83. package/native/shared/src/renderer/mod.rs +12556 -0
  84. package/native/shared/src/renderer/model_draw.rs +641 -0
  85. package/native/shared/src/renderer/occlusion.rs +429 -0
  86. package/native/shared/src/renderer/planar_pass.rs +593 -0
  87. package/native/shared/src/renderer/planar_reflection.rs +499 -0
  88. package/native/shared/src/renderer/post_pass.rs +249 -0
  89. package/native/shared/src/renderer/postfx_chain.rs +728 -0
  90. package/native/shared/src/renderer/pt_pass.rs +577 -0
  91. package/native/shared/src/renderer/scene_pass.rs +607 -0
  92. package/native/shared/src/renderer/shader_include.rs +205 -0
  93. package/native/shared/src/renderer/shader_library.rs +135 -0
  94. package/native/shared/src/renderer/shaders/ao.rs +570 -0
  95. package/native/shared/src/renderer/shaders/core.rs +1243 -0
  96. package/native/shared/src/renderer/shaders/env.rs +907 -0
  97. package/native/shared/src/renderer/shaders/gi.rs +810 -0
  98. package/native/shared/src/renderer/shaders/mod.rs +19 -0
  99. package/native/shared/src/renderer/shaders/post.rs +1558 -0
  100. package/native/shared/src/renderer/shaders/pt.rs +1859 -0
  101. package/native/shared/src/renderer/shaders/ssgi.rs +1586 -0
  102. package/native/shared/src/renderer/shadow_pass.rs +731 -0
  103. package/native/shared/src/renderer/ssgi_pass.rs +392 -0
  104. package/native/shared/src/renderer/ssr_pass.rs +188 -0
  105. package/native/shared/src/renderer/texture_store.rs +473 -0
  106. package/native/shared/src/renderer/transient.rs +591 -0
  107. package/native/shared/src/renderer/types.rs +941 -0
  108. package/native/shared/src/renderer/util.rs +152 -0
  109. package/native/shared/src/scene.rs +1362 -0
  110. package/native/shared/src/sdf_cache.rs +274 -0
  111. package/native/shared/src/shadows.rs +1036 -0
  112. package/native/shared/src/staging.rs +102 -0
  113. package/native/shared/src/string_header.rs +266 -0
  114. package/native/shared/src/text_renderer.rs +502 -0
  115. package/native/shared/src/textures.rs +197 -0
  116. package/native/tvos/Cargo.lock +1693 -0
  117. package/native/tvos/Cargo.toml +36 -0
  118. package/native/tvos/metal-patched/Cargo.toml +178 -0
  119. package/native/tvos/metal-patched/LICENSE-APACHE +201 -0
  120. package/native/tvos/metal-patched/LICENSE-MIT +25 -0
  121. package/native/tvos/metal-patched/src/acceleration_structure.rs +667 -0
  122. package/native/tvos/metal-patched/src/acceleration_structure_pass.rs +108 -0
  123. package/native/tvos/metal-patched/src/argument.rs +366 -0
  124. package/native/tvos/metal-patched/src/blitpass.rs +102 -0
  125. package/native/tvos/metal-patched/src/buffer.rs +71 -0
  126. package/native/tvos/metal-patched/src/capturedescriptor.rs +76 -0
  127. package/native/tvos/metal-patched/src/capturemanager.rs +113 -0
  128. package/native/tvos/metal-patched/src/commandbuffer.rs +192 -0
  129. package/native/tvos/metal-patched/src/commandqueue.rs +44 -0
  130. package/native/tvos/metal-patched/src/computepass.rs +107 -0
  131. package/native/tvos/metal-patched/src/constants.rs +152 -0
  132. package/native/tvos/metal-patched/src/counters.rs +119 -0
  133. package/native/tvos/metal-patched/src/depthstencil.rs +190 -0
  134. package/native/tvos/metal-patched/src/device.rs +2134 -0
  135. package/native/tvos/metal-patched/src/drawable.rs +39 -0
  136. package/native/tvos/metal-patched/src/encoder.rs +2041 -0
  137. package/native/tvos/metal-patched/src/heap.rs +281 -0
  138. package/native/tvos/metal-patched/src/indirect_encoder.rs +344 -0
  139. package/native/tvos/metal-patched/src/lib.rs +657 -0
  140. package/native/tvos/metal-patched/src/library.rs +902 -0
  141. package/native/tvos/metal-patched/src/mps.rs +575 -0
  142. package/native/tvos/metal-patched/src/pipeline/compute.rs +475 -0
  143. package/native/tvos/metal-patched/src/pipeline/mod.rs +71 -0
  144. package/native/tvos/metal-patched/src/pipeline/render.rs +762 -0
  145. package/native/tvos/metal-patched/src/renderpass.rs +443 -0
  146. package/native/tvos/metal-patched/src/resource.rs +182 -0
  147. package/native/tvos/metal-patched/src/sampler.rs +165 -0
  148. package/native/tvos/metal-patched/src/sync.rs +178 -0
  149. package/native/tvos/metal-patched/src/texture.rs +352 -0
  150. package/native/tvos/metal-patched/src/types.rs +90 -0
  151. package/native/tvos/metal-patched/src/vertexdescriptor.rs +250 -0
  152. package/native/tvos/src/audio_backend.rs +197 -0
  153. package/native/tvos/src/lib.rs +1891 -0
  154. package/native/visionos/Cargo.lock +1693 -0
  155. package/native/visionos/Cargo.toml +40 -0
  156. package/native/visionos/src/audio_backend.rs +197 -0
  157. package/native/visionos/src/lib.rs +1887 -0
  158. package/native/watchos/Cargo.lock +16 -0
  159. package/native/watchos/Cargo.toml +19 -0
  160. package/native/watchos/shaders/bloom_postfx.metal +99 -0
  161. package/native/watchos/src/BloomWatchApp.swift +1267 -0
  162. package/native/watchos/src/BloomWatchAudio.swift +179 -0
  163. package/native/watchos/src/audio.rs +55 -0
  164. package/native/watchos/src/draw_list.rs +229 -0
  165. package/native/watchos/src/ffi_stubs.rs +915 -0
  166. package/native/watchos/src/ffi_stubs_manual.rs +35 -0
  167. package/native/watchos/src/lib.rs +1124 -0
  168. package/native/watchos/src/models.rs +746 -0
  169. package/native/watchos/src/postfx.rs +95 -0
  170. package/native/watchos/src/scene.rs +534 -0
  171. package/native/watchos/src/textures.rs +184 -0
  172. package/native/web/Cargo.lock +1657 -0
  173. package/native/web/Cargo.toml +43 -0
  174. package/native/web/bloom_glue.js +695 -0
  175. package/native/web/build.sh +131 -0
  176. package/native/web/index.html +35 -0
  177. package/native/web/jolt_bridge.js +1519 -0
  178. package/native/web/src/input_ffi.rs +286 -0
  179. package/native/web/src/lib.rs +1796 -0
  180. package/native/web/src/material_ffi.rs +710 -0
  181. package/native/web/src/parity_ffi.rs +343 -0
  182. package/native/web/src/physics_ffi.rs +643 -0
  183. package/native/web/src/ragdoll_ffi.rs +250 -0
  184. package/native/web/src/render_settings.rs +98 -0
  185. package/native/windows/Cargo.lock +1815 -0
  186. package/native/windows/Cargo.toml +68 -0
  187. package/native/windows/src/lib.rs +1486 -0
  188. package/package.json +4279 -0
  189. package/src/audio/index.ts +315 -0
  190. package/src/core/colors.ts +63 -0
  191. package/src/core/index.ts +1206 -0
  192. package/src/core/keys.ts +63 -0
  193. package/src/core/types.ts +104 -0
  194. package/src/index.ts +171 -0
  195. package/src/math/index.ts +516 -0
  196. package/src/mobile/index.ts +294 -0
  197. package/src/models/index.ts +1258 -0
  198. package/src/physics/index.ts +1134 -0
  199. package/src/scene/index.ts +698 -0
  200. package/src/shapes/index.ts +120 -0
  201. package/src/text/index.ts +48 -0
  202. package/src/textures/index.ts +187 -0
  203. package/src/vfx/index.ts +191 -0
  204. package/src/world/index.ts +24 -0
  205. package/src/world/loader.ts +423 -0
  206. package/src/world/prefab.ts +217 -0
  207. package/src/world/render.ts +172 -0
  208. package/src/world/saver.ts +108 -0
  209. package/src/world/serialize.ts +301 -0
  210. package/src/world/terrain.ts +355 -0
  211. package/src/world/types.ts +160 -0
  212. package/src/world/validate.ts +319 -0
  213. package/src/world/version.ts +114 -0
@@ -0,0 +1,1558 @@
1
+ //! Post-processing: bloom, DoF, motion blur, SSS, compose, TAA, exposure, composite, upscale, RCAS.
2
+ //! Split from renderer/shaders.rs.
3
+
4
+ /// Bloom mip-chain shader. Three fragment entry points share a
5
+ /// single vertex stage and uniform layout:
6
+ ///
7
+ /// - `fs_downsample`: 4-tap box-filter downsample for mips ≥ 1.
8
+ /// - `fs_threshold_downsample`: same downsample but applies a
9
+ /// Karis-style soft threshold first to extract HDR brights and
10
+ /// suppress fireflies. Used only when sampling the source HDR
11
+ /// into bloom mip 0.
12
+ /// - `fs_upsample`: 9-tap tent-filter upsample, additive blend
13
+ /// (set via wgpu's blend state on the upsample pipeline).
14
+ ///
15
+ /// Uniform: `params.xy` = source texel size (1/src_w, 1/src_h);
16
+ /// `params.z` = bloom intensity (only used by upsample); `params.w`
17
+ /// reserved.
18
+ pub(in crate::renderer) const BLOOM_SHADER_WGSL: &str = "
19
+ struct BloomParams {
20
+ /// xy = source texel size (1/src_w, 1/src_h),
21
+ /// z = filter radius (upsample tent),
22
+ /// w = HDR threshold (downsample-threshold variant only).
23
+ params: vec4<f32>,
24
+ };
25
+
26
+ @group(0) @binding(0) var<uniform> u: BloomParams;
27
+ @group(0) @binding(1) var src_tex: texture_2d<f32>;
28
+ @group(0) @binding(2) var src_samp: sampler;
29
+
30
+ struct VsOut {
31
+ @builtin(position) clip_pos: vec4<f32>,
32
+ @location(0) uv: vec2<f32>,
33
+ };
34
+
35
+ @vertex
36
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
37
+ let x = f32((vid & 1u) * 4u) - 1.0;
38
+ let y = f32((vid >> 1u) * 4u) - 1.0;
39
+ var out: VsOut;
40
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
41
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
42
+ return out;
43
+ }
44
+
45
+ fn karis_average(c: vec4<f32>) -> vec4<f32> {
46
+ // 1 / (luma + 1) weighting to suppress fireflies — heavy bright
47
+ // pixels get downweighted so a single hot texel can't dominate
48
+ // a bloom kernel and create visible specks.
49
+ let luma = dot(c.rgb, vec3<f32>(0.2126, 0.7152, 0.0722));
50
+ let weight = 1.0 / (1.0 + luma);
51
+ return c * weight;
52
+ }
53
+
54
+ // Soft HDR threshold (UE-style). Pixels with luminance below
55
+ // `threshold - knee` get zero contribution; above `threshold + knee`
56
+ // pass through fully; in-between blends smoothly. Without this,
57
+ // bloom would brighten EVERY pixel in the scene rather than just
58
+ // the visibly-overbright ones (sun glare, emissive accents, etc.).
59
+ //
60
+ // The min() clamps the input to a large finite value — prevents
61
+ // Inf/NaN from PBR edge cases (grazing-angle divides in the scene
62
+ // shader, specular hotspots hitting Rgba16Float's max) from
63
+ // propagating through bloom and poisoning every downstream pass.
64
+ fn extract_brights(c_in: vec3<f32>, threshold: f32, knee: f32) -> vec3<f32> {
65
+ // NaN-safe clamp: max(.,0) first (flushes NaN→0 on most
66
+ // platforms), then cap at a large finite to avoid Inf.
67
+ let c = min(max(c_in, vec3<f32>(0.0)), vec3<f32>(64000.0));
68
+ let luma = dot(c, vec3<f32>(0.2126, 0.7152, 0.0722));
69
+ let lower = max(threshold - knee, 0.0);
70
+ let upper = threshold + knee;
71
+ let factor = smoothstep(lower, upper, luma);
72
+ return c * factor;
73
+ }
74
+
75
+ // 13-tap downsample (Sledgehammer / Karis 2013). Takes 5 dual-2x2
76
+ // box samples + 4 cross samples around each fragment for a smoother
77
+ // reduction than a naive 4-tap.
78
+ // Sanitize a sample: force NaN→0 via max-with-0, cap Inf via min.
79
+ // Needed because the HDR source can contain Inf from PBR edge cases
80
+ // (grazing-angle specular, division-by-zero-ish terms). Without
81
+ // this, one bad texel poisons the whole downsample kernel.
82
+ fn sanitize(c: vec4<f32>) -> vec4<f32> {
83
+ return vec4<f32>(min(max(c.rgb, vec3<f32>(0.0)), vec3<f32>(64000.0)), c.a);
84
+ }
85
+
86
+ fn downsample_13(uv: vec2<f32>, src_size: vec2<f32>, do_threshold: bool) -> vec3<f32> {
87
+ let dx = src_size.x;
88
+ let dy = src_size.y;
89
+
90
+ let a = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>(-2.0 * dx, -2.0 * dy)));
91
+ let b = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>( 0.0, -2.0 * dy)));
92
+ let c = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>( 2.0 * dx, -2.0 * dy)));
93
+ let d = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>(-2.0 * dx, 0.0)));
94
+ let e = sanitize(textureSample(src_tex, src_samp, uv));
95
+ let f = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>( 2.0 * dx, 0.0)));
96
+ let g = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>(-2.0 * dx, 2.0 * dy)));
97
+ let h = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>( 0.0, 2.0 * dy)));
98
+ let i = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>( 2.0 * dx, 2.0 * dy)));
99
+ let j = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>(-1.0 * dx, -1.0 * dy)));
100
+ let k = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>( 1.0 * dx, -1.0 * dy)));
101
+ let l = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>(-1.0 * dx, 1.0 * dy)));
102
+ let m = sanitize(textureSample(src_tex, src_samp, uv + vec2<f32>( 1.0 * dx, 1.0 * dy)));
103
+
104
+ // Five 2x2 boxes weighted to eliminate aliasing.
105
+ var groups: array<vec4<f32>, 5>;
106
+ groups[0] = (a + b + d + e) * 0.25;
107
+ groups[1] = (b + c + e + f) * 0.25;
108
+ groups[2] = (d + e + g + h) * 0.25;
109
+ groups[3] = (e + f + h + i) * 0.25;
110
+ groups[4] = (j + k + l + m) * 0.25;
111
+
112
+ if (do_threshold) {
113
+ // First extract HDR brights via soft threshold, then Karis
114
+ // weight to keep fireflies from poking through.
115
+ //
116
+ // Threshold arrives in params.w (pre-exposure HDR units),
117
+ // computed CPU-side as 2.5 / manual_exposure so bloom keeps
118
+ // triggering at the same DISPLAY brightness whatever the
119
+ // game's exposure (2.5 was tuned at Sponza's exposure 1.0;
120
+ // auto-exposure keeps the plain 2.5). Knee stays at the
121
+ // reference 2.5:0.5 ratio. History: the 'glitter on the
122
+ // floor' bug that briefly pushed the constant to 8.0 turned
123
+ // out to be the irradiance-convolution firefly leak (see
124
+ // fix_ibl commit), not bloom. At the reference exposure,
125
+ // diffuse sunlit stone (luma 2-3) sits right at the knee —
126
+ // barely blooming — while sky / emissive / specular peaks
127
+ // still get a proper halo.
128
+ let thr = max(u.params.w, 0.05);
129
+ let knee = thr * 0.2;
130
+ for (var n = 0u; n < 5u; n = n + 1u) {
131
+ let bright = extract_brights(groups[n].rgb, thr, knee);
132
+ let weighted = karis_average(vec4<f32>(bright, 1.0));
133
+ groups[n] = weighted;
134
+ }
135
+ }
136
+
137
+ let weights = array<f32, 5>(0.125, 0.125, 0.125, 0.125, 0.5);
138
+ var sum = vec4<f32>(0.0);
139
+ for (var n = 0u; n < 5u; n = n + 1u) {
140
+ sum = sum + groups[n] * weights[n];
141
+ }
142
+ return sum.rgb;
143
+ }
144
+
145
+ @fragment
146
+ fn fs_downsample(in: VsOut) -> @location(0) vec4<f32> {
147
+ return vec4<f32>(downsample_13(in.uv, u.params.xy, false), 1.0);
148
+ }
149
+
150
+ @fragment
151
+ fn fs_threshold_downsample(in: VsOut) -> @location(0) vec4<f32> {
152
+ return vec4<f32>(downsample_13(in.uv, u.params.xy, true), 1.0);
153
+ }
154
+
155
+ // 9-tap tent filter upsample (Sledgehammer). Texel-radius scaled by
156
+ // the small radius factor in u.params.z (defaults to ~1.0 — wider
157
+ // = more blurry overlap). Output is BLENDED additively into the
158
+ // destination via the upsample pipeline's blend state.
159
+ @fragment
160
+ fn fs_upsample(in: VsOut) -> @location(0) vec4<f32> {
161
+ let dx = u.params.x * u.params.z;
162
+ let dy = u.params.y * u.params.z;
163
+ let uv = in.uv;
164
+
165
+ var sum = vec3<f32>(0.0);
166
+ sum = sum + textureSample(src_tex, src_samp, uv + vec2<f32>(-dx, dy)).rgb * 1.0;
167
+ sum = sum + textureSample(src_tex, src_samp, uv + vec2<f32>( 0.0, dy)).rgb * 2.0;
168
+ sum = sum + textureSample(src_tex, src_samp, uv + vec2<f32>( dx, dy)).rgb * 1.0;
169
+
170
+ sum = sum + textureSample(src_tex, src_samp, uv + vec2<f32>(-dx, 0.0)).rgb * 2.0;
171
+ sum = sum + textureSample(src_tex, src_samp, uv).rgb * 4.0;
172
+ sum = sum + textureSample(src_tex, src_samp, uv + vec2<f32>( dx, 0.0)).rgb * 2.0;
173
+
174
+ sum = sum + textureSample(src_tex, src_samp, uv + vec2<f32>(-dx, -dy)).rgb * 1.0;
175
+ sum = sum + textureSample(src_tex, src_samp, uv + vec2<f32>( 0.0, -dy)).rgb * 2.0;
176
+ sum = sum + textureSample(src_tex, src_samp, uv + vec2<f32>( dx, -dy)).rgb * 1.0;
177
+
178
+ return vec4<f32>(sum * (1.0 / 16.0), 1.0);
179
+ }
180
+ ";
181
+
182
+ pub(in crate::renderer) const DOF_SHADER_WGSL: &str = "
183
+ struct DofParams {
184
+ params: vec4<f32>, // x = focus_distance (view-space Z, positive), y = aperture (CoC scale), z = max_blur_radius (UV), w = unused
185
+ inv_proj: mat4x4<f32>,
186
+ };
187
+
188
+ @group(0) @binding(0) var<uniform> u: DofParams;
189
+ @group(0) @binding(1) var color_tex: texture_2d<f32>;
190
+ @group(0) @binding(2) var color_samp: sampler;
191
+ @group(0) @binding(3) var depth_tex: texture_depth_2d;
192
+ @group(0) @binding(4) var depth_samp: sampler;
193
+
194
+ struct VsOut {
195
+ @builtin(position) clip_pos: vec4<f32>,
196
+ @location(0) uv: vec2<f32>,
197
+ };
198
+
199
+ @vertex
200
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
201
+ let x = f32((vid & 1u) * 4u) - 1.0;
202
+ let y = f32((vid >> 1u) * 4u) - 1.0;
203
+ var out: VsOut;
204
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
205
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
206
+ return out;
207
+ }
208
+
209
+ // Reconstruct view-space Z from depth buffer value via inverse projection.
210
+ fn linearize_depth(depth: f32) -> f32 {
211
+ let ndc = vec4<f32>(0.0, 0.0, depth, 1.0);
212
+ let vp = u.inv_proj * ndc;
213
+ return -vp.z / vp.w; // positive distance from camera
214
+ }
215
+
216
+ // 16-sample Poisson disc (same offsets used by shadow PCF).
217
+ const POISSON_16: array<vec2<f32>, 16> = array<vec2<f32>, 16>(
218
+ vec2<f32>(-0.94201624, -0.39906216),
219
+ vec2<f32>( 0.94558609, -0.76890725),
220
+ vec2<f32>(-0.09418410, -0.92938870),
221
+ vec2<f32>( 0.34495938, 0.29387760),
222
+ vec2<f32>(-0.91588581, 0.45771432),
223
+ vec2<f32>(-0.81544232, -0.87912464),
224
+ vec2<f32>(-0.38277543, 0.27676845),
225
+ vec2<f32>( 0.97484398, 0.75648379),
226
+ vec2<f32>( 0.44323325, -0.97511554),
227
+ vec2<f32>( 0.53742981, -0.47373420),
228
+ vec2<f32>(-0.26496911, -0.41893023),
229
+ vec2<f32>( 0.79197514, 0.19090188),
230
+ vec2<f32>(-0.24188840, 0.99706507),
231
+ vec2<f32>(-0.81409955, 0.91437590),
232
+ vec2<f32>( 0.19984126, 0.78641367),
233
+ vec2<f32>( 0.14383161, -0.14100790),
234
+ );
235
+
236
+ @fragment
237
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
238
+ let focus_dist = u.params.x;
239
+ let aperture = u.params.y;
240
+ let max_blur = u.params.z;
241
+
242
+ let center_depth_raw = textureSampleLevel(depth_tex, depth_samp, in.uv, 0u);
243
+
244
+ // Sky pixels (depth ~1.0) get maximum blur — they are at infinity.
245
+ var view_z: f32;
246
+ if (center_depth_raw >= 0.9999) {
247
+ view_z = 1000.0; // treat as very far
248
+ } else {
249
+ view_z = linearize_depth(center_depth_raw);
250
+ }
251
+
252
+ // Circle of confusion: thin-lens approximation.
253
+ // Dividing by view_z ensures distant objects don't get disproportionately
254
+ // blurred — CoC grows with defocus distance but falls off with depth.
255
+ // max(view_z, 0.1) prevents division by zero for geometry very close to
256
+ // the camera.
257
+ let coc = clamp(aperture * abs(view_z - focus_dist) / max(view_z, 0.1), 0.0, max_blur);
258
+
259
+ // If CoC is negligibly small, return the source pixel unchanged.
260
+ let threshold = 0.0005;
261
+ if (coc < threshold) {
262
+ return textureSampleLevel(color_tex, color_samp, in.uv, 0.0);
263
+ }
264
+
265
+ // Gather 16 Poisson disc samples scaled by CoC.
266
+ var color_sum = vec3<f32>(0.0);
267
+ var weight_sum = 0.0;
268
+
269
+ let center_color = textureSampleLevel(color_tex, color_samp, in.uv, 0.0).rgb;
270
+
271
+ for (var i = 0u; i < 16u; i = i + 1u) {
272
+ let offset = POISSON_16[i] * coc;
273
+ let sample_uv = in.uv + offset;
274
+
275
+ let sample_color = textureSampleLevel(color_tex, color_samp, sample_uv, 0.0).rgb;
276
+
277
+ // Read the depth at the sample location to compute its own CoC.
278
+ // This prevents sharp foreground objects from bleeding into
279
+ // blurred background — only samples that are themselves blurry
280
+ // (or at least as blurry as this pixel) contribute fully.
281
+ let sample_depth_raw = textureSampleLevel(depth_tex, depth_samp, sample_uv, 0u);
282
+ var sample_z: f32;
283
+ if (sample_depth_raw >= 0.9999) {
284
+ sample_z = 1000.0;
285
+ } else {
286
+ sample_z = linearize_depth(sample_depth_raw);
287
+ }
288
+ let sample_coc = clamp(abs(sample_z - focus_dist) * aperture, 0.0, max_blur);
289
+
290
+ // Weight: accept the sample if its CoC is at least as large as
291
+ // the center pixel's CoC, or if the sample is behind the center
292
+ // (background blurring into foreground is expected). Otherwise
293
+ // attenuate by the ratio of sample_coc / coc.
294
+ var w = 1.0;
295
+ if (sample_z < view_z) {
296
+ // Sample is in front of center — only contribute if it is
297
+ // itself blurry enough.
298
+ w = saturate(sample_coc / coc);
299
+ }
300
+
301
+ color_sum += sample_color * w;
302
+ weight_sum += w;
303
+ }
304
+
305
+ // Also blend in the center pixel with weight 1.
306
+ color_sum += center_color;
307
+ weight_sum += 1.0;
308
+
309
+ let result = color_sum / weight_sum;
310
+ return vec4<f32>(result, 1.0);
311
+ }
312
+ ";
313
+
314
+ pub(in crate::renderer) const MOTION_BLUR_SHADER_WGSL: &str = "
315
+ struct MotionBlurParams {
316
+ /// x = strength (velocity multiplier), y = max_blur (UV clamp), zw = unused.
317
+ params: vec4<f32>,
318
+ };
319
+
320
+ @group(0) @binding(0) var<uniform> u: MotionBlurParams;
321
+ @group(0) @binding(1) var color_tex: texture_2d<f32>;
322
+ @group(0) @binding(2) var color_samp: sampler;
323
+ @group(0) @binding(3) var velocity_tex: texture_2d<f32>;
324
+ @group(0) @binding(4) var velocity_samp: sampler;
325
+
326
+ struct VsOut {
327
+ @builtin(position) clip_pos: vec4<f32>,
328
+ @location(0) uv: vec2<f32>,
329
+ };
330
+
331
+ @vertex
332
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
333
+ let x = f32((vid & 1u) * 4u) - 1.0;
334
+ let y = f32((vid >> 1u) * 4u) - 1.0;
335
+ var out: VsOut;
336
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
337
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
338
+ return out;
339
+ }
340
+
341
+ @fragment
342
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
343
+ let strength = u.params.x;
344
+ let max_blur = u.params.y;
345
+
346
+ let vel_raw = textureSampleLevel(velocity_tex, velocity_samp, in.uv, 0.0).rg;
347
+ // Scale velocity by strength and clamp to max_blur radius.
348
+ var vel = vel_raw * strength;
349
+ let vel_len = length(vel);
350
+ if (vel_len > max_blur) {
351
+ vel = vel * (max_blur / vel_len);
352
+ }
353
+
354
+ // If velocity is negligible, return source pixel unchanged.
355
+ if (vel_len < 0.0001) {
356
+ return textureSampleLevel(color_tex, color_samp, in.uv, 0.0);
357
+ }
358
+
359
+ // 8-tap directional blur with tent (linear) weighting.
360
+ // Samples are placed symmetrically around the center pixel
361
+ // along the velocity vector. The tent filter peaks at the
362
+ // center and falls off linearly toward the endpoints.
363
+ let n_samples: i32 = 8;
364
+ var color_sum = vec3<f32>(0.0);
365
+ var weight_sum = 0.0;
366
+ for (var i: i32 = 0; i < n_samples; i = i + 1) {
367
+ let t = (f32(i) + 0.5) / f32(n_samples) - 0.5; // range [-0.5, 0.5)
368
+ let sample_uv = in.uv + vel * t;
369
+ let w = 1.0 - abs(t * 2.0); // tent weight: 1.0 at center, 0 at edges
370
+ color_sum += textureSampleLevel(color_tex, color_samp, sample_uv, 0.0).rgb * w;
371
+ weight_sum += w;
372
+ }
373
+
374
+ return vec4<f32>(color_sum / weight_sum, 1.0);
375
+ }
376
+ ";
377
+
378
+ pub(in crate::renderer) const SSS_SHADER_WGSL: &str = "
379
+ struct SssParams {
380
+ /// x = strength (0 = off, 1 = full blend), y = width (screen-space
381
+ /// blur radius in UV units, e.g. 0.01), z = falloff (bilateral
382
+ /// depth edge-stop steepness), w = unused.
383
+ params: vec4<f32>,
384
+ };
385
+
386
+ @group(0) @binding(0) var<uniform> u: SssParams;
387
+ @group(0) @binding(1) var color_tex: texture_2d<f32>;
388
+ @group(0) @binding(2) var color_samp: sampler;
389
+ @group(0) @binding(3) var depth_tex: texture_depth_2d;
390
+ @group(0) @binding(4) var depth_samp: sampler;
391
+
392
+ struct VsOut {
393
+ @builtin(position) clip_pos: vec4<f32>,
394
+ @location(0) uv: vec2<f32>,
395
+ };
396
+
397
+ @vertex
398
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
399
+ let x = f32((vid & 1u) * 4u) - 1.0;
400
+ let y = f32((vid >> 1u) * 4u) - 1.0;
401
+ var out: VsOut;
402
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
403
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
404
+ return out;
405
+ }
406
+
407
+ // 9-tap disc pattern (unit disc, slightly stratified).
408
+ // Kept intentionally modest — SSS scatter radius is small.
409
+ const DISC_9: array<vec2<f32>, 9> = array<vec2<f32>, 9>(
410
+ vec2<f32>( 0.0, 0.0),
411
+ vec2<f32>( 1.0, 0.0),
412
+ vec2<f32>(-1.0, 0.0),
413
+ vec2<f32>( 0.0, 1.0),
414
+ vec2<f32>( 0.0, -1.0),
415
+ vec2<f32>( 0.7071, 0.7071),
416
+ vec2<f32>(-0.7071, 0.7071),
417
+ vec2<f32>( 0.7071, -0.7071),
418
+ vec2<f32>(-0.7071, -0.7071),
419
+ );
420
+
421
+ @fragment
422
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
423
+ let strength = u.params.x;
424
+ let width = u.params.y;
425
+ let falloff = u.params.z;
426
+
427
+ let center_color = textureSampleLevel(color_tex, color_samp, in.uv, 0.0);
428
+
429
+ // Sky pixels (raw depth ~1.0) skip SSS entirely — they have no
430
+ // geometry to scatter through. This also avoids depth-edge halos
431
+ // at the horizon.
432
+ let center_depth = textureSampleLevel(depth_tex, depth_samp, in.uv, 0u);
433
+ if (center_depth >= 0.9999) {
434
+ return center_color;
435
+ }
436
+
437
+ // Chromatic diffusion profile: red scatters furthest (skin
438
+ // absorbs blue/green more than red). Width multipliers:
439
+ // red = 1.0 × width
440
+ // green = 0.5 × width
441
+ // blue = 0.25 × width
442
+ var sum_r = 0.0;
443
+ var sum_g = 0.0;
444
+ var sum_b = 0.0;
445
+ var weight_r = 0.0;
446
+ var weight_g = 0.0;
447
+ var weight_b = 0.0;
448
+
449
+ for (var i = 0u; i < 9u; i = i + 1u) {
450
+ let tap_r = in.uv + DISC_9[i] * width;
451
+ let tap_g = in.uv + DISC_9[i] * (width * 0.5);
452
+ let tap_b = in.uv + DISC_9[i] * (width * 0.25);
453
+
454
+ // Bilateral depth weight — each channel uses its own tap UV,
455
+ // so we sample depth at each channel's location independently.
456
+ let d_r = textureSampleLevel(depth_tex, depth_samp, tap_r, 0u);
457
+ let d_g = textureSampleLevel(depth_tex, depth_samp, tap_g, 0u);
458
+ let d_b = textureSampleLevel(depth_tex, depth_samp, tap_b, 0u);
459
+
460
+ let w_r = exp(-abs(d_r - center_depth) * falloff);
461
+ let w_g = exp(-abs(d_g - center_depth) * falloff);
462
+ let w_b = exp(-abs(d_b - center_depth) * falloff);
463
+
464
+ // Spatial Gaussian (unit disc → standard Gaussian weight from
465
+ // the squared distance within the disc).
466
+ let dist2 = dot(DISC_9[i], DISC_9[i]);
467
+ let gauss = exp(-dist2 * 2.0); // sigma ≈ 0.7 in disc-space
468
+
469
+ let c_r = textureSampleLevel(color_tex, color_samp, tap_r, 0.0).r;
470
+ let c_g = textureSampleLevel(color_tex, color_samp, tap_g, 0.0).g;
471
+ let c_b = textureSampleLevel(color_tex, color_samp, tap_b, 0.0).b;
472
+
473
+ sum_r += c_r * w_r * gauss;
474
+ sum_g += c_g * w_g * gauss;
475
+ sum_b += c_b * w_b * gauss;
476
+ weight_r += w_r * gauss;
477
+ weight_g += w_g * gauss;
478
+ weight_b += w_b * gauss;
479
+ }
480
+
481
+ let blurred = vec3<f32>(
482
+ sum_r / max(weight_r, 1e-5),
483
+ sum_g / max(weight_g, 1e-5),
484
+ sum_b / max(weight_b, 1e-5),
485
+ );
486
+
487
+ // Blend blurred result with original by strength.
488
+ let result = mix(center_color.rgb, blurred, strength);
489
+ return vec4<f32>(result, center_color.a);
490
+ }
491
+ ";
492
+
493
+ /// Scene-compose shader. Merges direct scene HDR with every
494
+ /// screen-space post effect (SSR, albedo-modulated SSGI, bloom) and
495
+ /// then applies volumetric fog + sun shafts. The output is a single
496
+ /// "composed HDR" texture that downstream passes consume:
497
+ ///
498
+ /// - TAA-on: the TAA pass reads composed_rt as its current frame
499
+ /// and only performs temporal reprojection + neighborhood clamp.
500
+ /// - TAA-off: the composite pass reads composed_rt directly.
501
+ ///
502
+ /// This keeps fog / shafts consistent across both TAA states and
503
+ /// removes the need for TAA / composite to re-compose the same
504
+ /// ingredients separately.
505
+ pub(in crate::renderer) const SCENE_COMPOSE_SHADER_WGSL: &str = "
506
+ struct SceneComposeParams {
507
+ /// x = bloom intensity; y/z/w padding.
508
+ misc: vec4<f32>,
509
+ /// Inverse of the current-frame view-projection (world-pos reconstruction).
510
+ inv_vp: mat4x4<f32>,
511
+ /// Fog tint (rgb) + density (w).
512
+ fog_color_density: vec4<f32>,
513
+ /// Fog: x = height_ref, y = falloff rate, zw padding.
514
+ fog_params: vec4<f32>,
515
+ /// Sun shafts: xy = projected sun UV, z = strength, w = decay.
516
+ sun_shaft_uv_strength: vec4<f32>,
517
+ /// Sun shaft tint (rgb, w padding).
518
+ sun_shaft_color: vec4<f32>,
519
+ };
520
+
521
+ @group(0) @binding(0) var<uniform> u: SceneComposeParams;
522
+ @group(0) @binding(1) var hdr_tex: texture_2d<f32>;
523
+ @group(0) @binding(2) var hdr_samp: sampler;
524
+ @group(0) @binding(3) var ssr_tex: texture_2d<f32>;
525
+ @group(0) @binding(4) var ssr_samp: sampler;
526
+ @group(0) @binding(5) var ssgi_tex: texture_2d<f32>;
527
+ @group(0) @binding(6) var ssgi_samp: sampler;
528
+ @group(0) @binding(7) var bloom_tex: texture_2d<f32>;
529
+ @group(0) @binding(8) var bloom_samp: sampler;
530
+ @group(0) @binding(9) var albedo_tex: texture_2d<f32>;
531
+ @group(0) @binding(10) var albedo_samp: sampler;
532
+ @group(0) @binding(11) var depth_tex: texture_depth_2d;
533
+ @group(0) @binding(12) var depth_samp: sampler;
534
+ @group(0) @binding(13) var aerial_tex: texture_3d<f32>;
535
+ @group(0) @binding(14) var aerial_samp: sampler;
536
+
537
+ struct VsOut {
538
+ @builtin(position) clip_pos: vec4<f32>,
539
+ @location(0) uv: vec2<f32>,
540
+ };
541
+
542
+ @vertex
543
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
544
+ let x = f32((vid & 1u) * 4u) - 1.0;
545
+ let y = f32((vid >> 1u) * 4u) - 1.0;
546
+ var out: VsOut;
547
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
548
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
549
+ return out;
550
+ }
551
+
552
+ @fragment
553
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
554
+ // Pre-tonemap HDR composition. SSR is already Fresnel/edge faded
555
+ // at its pass; SSGI is raw indirect radiance, multiplied here by
556
+ // the receiver albedo so dark materials absorb correctly. Bloom
557
+ // is scaled by the user-tuned intensity.
558
+ let hdr = textureSampleLevel(hdr_tex, hdr_samp, in.uv, 0.0).rgb;
559
+ let ssr = textureSampleLevel(ssr_tex, ssr_samp, in.uv, 0.0).rgb;
560
+ let ssgi = textureSampleLevel(ssgi_tex, ssgi_samp, in.uv, 0.0).rgb;
561
+ let albedo_sample = textureSampleLevel(albedo_tex, albedo_samp, in.uv, 0.0);
562
+ let albedo = albedo_sample.rgb;
563
+ // albedo.a carries `1 - shadow_factor` from the scene pass — how
564
+ // much of this pixel's illumination is indirect (IBL + bounce) vs.
565
+ // direct (sun). Forwarded through the composed RT's alpha channel
566
+ // so the composite/tonemap pass can apply SSAO only to the
567
+ // indirect-dominated portion. Sky pixels carry 0 here so fog / AO
568
+ // don't touch them.
569
+ let indirect_weight = albedo_sample.a;
570
+ let bloom = textureSampleLevel(bloom_tex, bloom_samp, in.uv, 0.0).rgb;
571
+ var color = hdr + ssr + ssgi * albedo + bloom * u.misc.x;
572
+
573
+ // World-space position from depth for fog ray march.
574
+ let depth = textureSampleLevel(depth_tex, depth_samp, in.uv, 0u);
575
+ let ndc = vec4<f32>(in.uv.x * 2.0 - 1.0, (1.0 - in.uv.y) * 2.0 - 1.0, depth, 1.0);
576
+ let world_h = u.inv_vp * ndc;
577
+ let world = world_h.xyz / world_h.w;
578
+
579
+ // EN-005 V2 — when procedural sky is on (misc.y > 0), sample the
580
+ // pre-baked aerial-perspective 3D LUT instead of running the
581
+ // Beer-Lambert march. The LUT is indexed by (NDC.xy, depth-slice)
582
+ // where depth-slice = world_distance / max_dist_km.
583
+ if (u.misc.y > 0.5) {
584
+ let cam_pos = vec3<f32>(
585
+ u.inv_vp[3][0] / u.inv_vp[3][3],
586
+ u.inv_vp[3][1] / u.inv_vp[3][3],
587
+ u.inv_vp[3][2] / u.inv_vp[3][3],
588
+ );
589
+ // Engine units are metres; LUT covers misc.z km. Skip far-
590
+ // plane / sky pixels: depth == 1.0 means no scene geometry,
591
+ // and the procedural sky pass already drew the right colour
592
+ // for that pixel — fogging it again would double-tint.
593
+ if (depth < 1.0) {
594
+ let dist_m = length(world - cam_pos);
595
+ let dist_km = dist_m * 0.001;
596
+ let max_km = u.misc.z;
597
+ let depth_slice = clamp(dist_km / max_km, 0.0, 1.0);
598
+ let aerial = textureSampleLevel(
599
+ aerial_tex,
600
+ aerial_samp,
601
+ vec3<f32>(in.uv.x, in.uv.y, depth_slice),
602
+ 0.0,
603
+ );
604
+ let in_scatter = aerial.rgb;
605
+ let mean_t = aerial.a;
606
+ color = color * mean_t + in_scatter;
607
+ }
608
+ }
609
+ // Manual height fog. Runs additively *after* the procedural-sky aerial
610
+ // perspective (not just as its `else`), because that LUT is km-scaled and
611
+ // contributes ~nothing over a small (tens-of-metres) arena. This march is
612
+ // world-scale, so a low density gives controllable near-ground haze that
613
+ // adds aerial depth and softens the distant terrain edge. Skips sky pixels
614
+ // (depth == 1.0) so the procedural sky isn't double-tinted/washed.
615
+ {
616
+ let fog_density = u.fog_color_density.w;
617
+ if (fog_density > 0.0 && depth < 1.0) {
618
+ let height_ref = u.fog_params.x;
619
+ let height_falloff = u.fog_params.y;
620
+ let cam_pos = vec3<f32>(
621
+ u.inv_vp[3][0] / u.inv_vp[3][3],
622
+ u.inv_vp[3][1] / u.inv_vp[3][3],
623
+ u.inv_vp[3][2] / u.inv_vp[3][3],
624
+ );
625
+ let ray = world - cam_pos;
626
+ let dist = length(ray);
627
+ let ray_dir = ray / max(dist, 0.001);
628
+
629
+ let n_steps = 16u;
630
+ let step_size = dist / f32(n_steps);
631
+ var transmittance = 1.0;
632
+ var in_scatter = vec3<f32>(0.0);
633
+ for (var i = 0u; i < n_steps; i = i + 1u) {
634
+ let t = (f32(i) + 0.5) * step_size;
635
+ let p = cam_pos + ray_dir * t;
636
+ let height_fade = exp(-height_falloff * max(p.y - height_ref, 0.0));
637
+ let local_density = fog_density * height_fade;
638
+ let step_extinction = exp(-local_density * step_size);
639
+ in_scatter += u.fog_color_density.rgb * local_density * step_size * transmittance;
640
+ transmittance *= step_extinction;
641
+ }
642
+ color = color * transmittance + in_scatter;
643
+ }
644
+ }
645
+
646
+ // Sun shafts: 32-tap march from the pixel toward the projected
647
+ // sun UV, accumulating sky-visibility with per-sample decay.
648
+ // Strength 0 disables.
649
+ let shaft_strength = u.sun_shaft_uv_strength.z;
650
+ if (shaft_strength > 0.0) {
651
+ let sun_uv = u.sun_shaft_uv_strength.xy;
652
+ let decay = u.sun_shaft_uv_strength.w;
653
+ let n_samples: i32 = 32;
654
+ let delta = (sun_uv - in.uv) / f32(n_samples);
655
+ var pos = in.uv;
656
+ var weight = 1.0;
657
+ var accum = 0.0;
658
+ for (var i: i32 = 0; i < n_samples; i = i + 1) {
659
+ pos = pos + delta;
660
+ if (pos.x < 0.0 || pos.x > 1.0 || pos.y < 0.0 || pos.y > 1.0) {
661
+ continue;
662
+ }
663
+ let d = textureSampleLevel(depth_tex, depth_samp, pos, 0u);
664
+ let sky = smoothstep(0.998, 1.0, d);
665
+ accum = accum + sky * weight;
666
+ weight = weight * decay;
667
+ }
668
+ let norm = accum / f32(n_samples);
669
+ color = color + u.sun_shaft_color.rgb * norm * shaft_strength;
670
+ }
671
+
672
+ // Compose-wide NaN scrub + defensive luma cap. The actual
673
+ // stone-floor speckle was fixed upstream at the irradiance
674
+ // convolution shader (sun disc was leaking into the 'diffuse'
675
+ // map); this cap stays as a safety net against future
676
+ // over-bright contributors (rare ssr/ssgi anomaly + bloom).
677
+ // 50 is high enough to never clip normal scene brightness.
678
+ let color_clean = select(vec3<f32>(0.0), color, color == color);
679
+ let color_luma = dot(color_clean, vec3<f32>(0.2126, 0.7152, 0.0722));
680
+ let compose_cap = 50.0;
681
+ let color_scale = select(1.0, compose_cap / color_luma, color_luma > compose_cap);
682
+ return vec4<f32>(color_clean * color_scale, indirect_weight);
683
+ }
684
+ ";
685
+
686
+ /// TAA shader. Reads `composed_rt` (scene HDR + post-effects + fog +
687
+ /// shafts already merged upstream) and performs only temporal
688
+ /// reprojection with neighborhood clamp, blending against the
689
+ /// history RT. For static scenes the blend converges in ~10 frames
690
+ /// to a fully sub-pixel-resolved image.
691
+ pub(in crate::renderer) const TAA_SHADER_WGSL: &str = "
692
+ struct TaaParams {
693
+ /// x = blend factor (current-frame weight); yz = the CURRENT frame's
694
+ /// jitter as a composed-texture UV offset (see the unjitter note at the
695
+ /// current-frame sample); w padding.
696
+ params: vec4<f32>,
697
+ /// Inverse of the current-frame view-projection matrix —
698
+ /// reconstructs world-space position for history reprojection.
699
+ inv_vp: mat4x4<f32>,
700
+ /// Previous-frame view-projection — projects world pos into
701
+ /// history UV.
702
+ prev_vp: mat4x4<f32>,
703
+ };
704
+
705
+ @group(0) @binding(0) var<uniform> u: TaaParams;
706
+ @group(0) @binding(1) var composed_tex: texture_2d<f32>;
707
+ @group(0) @binding(2) var composed_samp: sampler;
708
+ @group(0) @binding(3) var history_tex: texture_2d<f32>;
709
+ @group(0) @binding(4) var history_samp: sampler;
710
+ @group(0) @binding(5) var depth_tex: texture_depth_2d;
711
+ @group(0) @binding(6) var depth_samp: sampler;
712
+ @group(0) @binding(7) var velocity_tex: texture_2d<f32>;
713
+ @group(0) @binding(8) var velocity_samp: sampler;
714
+
715
+ struct VsOut {
716
+ @builtin(position) clip_pos: vec4<f32>,
717
+ @location(0) uv: vec2<f32>,
718
+ };
719
+
720
+ @vertex
721
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
722
+ let x = f32((vid & 1u) * 4u) - 1.0;
723
+ let y = f32((vid >> 1u) * 4u) - 1.0;
724
+ var out: VsOut;
725
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
726
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
727
+ return out;
728
+ }
729
+
730
+ // RGB <-> YCoCg conversions. Reversible, linear, cheap (no matrix
731
+ // multiply). Used by the TAA neighborhood clamp so we can bound
732
+ // history's luma (Y) against the source neighborhood's statistical
733
+ // range while leaving chroma (Co, Cg) alone — the per-channel RGB
734
+ // clamp was causing chromatic sparkle on grazing-angle stone.
735
+ fn rgb_to_ycocg(c: vec3<f32>) -> vec3<f32> {
736
+ let Co = c.r - c.b;
737
+ let tmp = c.b + Co * 0.5;
738
+ let Cg = c.g - tmp;
739
+ let Y = tmp + Cg * 0.5;
740
+ return vec3<f32>(Y, Co, Cg);
741
+ }
742
+ fn ycocg_to_rgb(c: vec3<f32>) -> vec3<f32> {
743
+ let tmp = c.x - c.z * 0.5;
744
+ let g = c.z + tmp;
745
+ let b = tmp - c.y * 0.5;
746
+ let r = c.y + b;
747
+ return vec3<f32>(r, g, b);
748
+ }
749
+
750
+ // 5-tap Catmull-Rom upsample (Karis formulation). When the source
751
+ // (composed_tex) is half-res relative to the destination, naive
752
+ // bilinear loses sharpness; Catmull-Rom reconstructs a cubic-Hermite
753
+ // curve through 4 source taps which preserves edges. Costs 5 bilinear
754
+ // fetches vs 1 — worth it for the TSR upscale because the alternative
755
+ // is a perceptibly blurrier image.
756
+ fn sample_catmull_rom(uv: vec2<f32>) -> vec4<f32> {
757
+ let tex_size = vec2<f32>(textureDimensions(composed_tex));
758
+ let inv_size = 1.0 / tex_size;
759
+ let sample_pos = uv * tex_size;
760
+ let tex_pos1 = floor(sample_pos - 0.5) + 0.5;
761
+ let f = sample_pos - tex_pos1;
762
+
763
+ let w0 = f * (-0.5 + f * (1.0 - 0.5 * f));
764
+ let w1 = 1.0 + f * f * (-2.5 + 1.5 * f);
765
+ let w2 = f * (0.5 + f * (2.0 - 1.5 * f));
766
+ let w3 = f * f * (-0.5 + 0.5 * f);
767
+ let w12 = w1 + w2;
768
+ let offset12 = w2 / w12;
769
+
770
+ let tp0 = (tex_pos1 - 1.0) * inv_size;
771
+ let tp3 = (tex_pos1 + 2.0) * inv_size;
772
+ let tp12 = (tex_pos1 + offset12) * inv_size;
773
+
774
+ var result = vec4<f32>(0.0);
775
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp12.x, tp0.y), 0.0) * w12.x * w0.y;
776
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp0.x, tp12.y), 0.0) * w0.x * w12.y;
777
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp12.x, tp12.y), 0.0) * w12.x * w12.y;
778
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp3.x, tp12.y), 0.0) * w3.x * w12.y;
779
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp12.x, tp3.y), 0.0) * w12.x * w3.y;
780
+ return max(result, vec4<f32>(0.0));
781
+ }
782
+
783
+ @fragment
784
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
785
+ // composed_tex already carries HDR + SSR + SSGI*albedo + bloom +
786
+ // fog + shafts — TAA only needs to reproject history and blend.
787
+ // Alpha carries `indirect_weight` (see scene_compose) which the
788
+ // composite pass reads to apply AO only to indirect-dominated
789
+ // pixels; pass it through blended with the colour so history
790
+ // stays consistent.
791
+ // UNJITTER the current-frame taps (2026-07-16 sparkle slice 2). The
792
+ // composed image was RENDERED with this frame's sub-pixel jitter, so a
793
+ // feature that belongs at pixel p sits at p + jitter — sampling at in.uv
794
+ // fed alpha x (jitter-phase difference) of shimmer into every output
795
+ // frame by construction, which the variance rig measured as TAA adding
796
+ // 2x flicker on static detail versus TAA off. Resampling the current
797
+ // frame at its true (unjittered) position is how the jitter becomes
798
+ // sub-pixel INFORMATION (via the Catmull-Rom fractional reconstruction)
799
+ // instead of output noise. History reprojection stays in jittered space
800
+ // on purpose: the velocity-ref VP already bakes the current jitter
801
+ // (EN-022), so prev_uv lands correctly without this offset.
802
+ let src_uv = in.uv + u.params.yz;
803
+ let current_sample = sample_catmull_rom(src_uv);
804
+ let current = current_sample.rgb;
805
+ let current_w = current_sample.a;
806
+
807
+ let depth = textureSampleLevel(depth_tex, depth_samp, in.uv, 0u);
808
+ let ndc = vec4<f32>(in.uv.x * 2.0 - 1.0, (1.0 - in.uv.y) * 2.0 - 1.0, depth, 1.0);
809
+ let world_h = u.inv_vp * ndc;
810
+ let world = world_h.xyz / world_h.w;
811
+
812
+ let vel = textureSampleLevel(velocity_tex, velocity_samp, in.uv, 0.0).rg;
813
+ let vel_len = length(vel);
814
+ var prev_uv: vec2<f32>;
815
+ if (vel_len > 0.00001) {
816
+ prev_uv = vec2<f32>(in.uv.x - vel.x, in.uv.y + vel.y);
817
+ } else if (depth >= 0.9999) {
818
+ // Sky / far plane: the positional reconstruction divides by a
819
+ // near-zero w and reprojects sky pixels onto arbitrary scene
820
+ // points — the luma-only history clamp then locks that wrong
821
+ // chroma in forever (uniform green/red sky tint). The sky is at
822
+ // infinity, so reproject the view DIRECTION instead: exact under
823
+ // camera rotation, translation-invariant by definition.
824
+ let dir = world_h.xyz; // w ~ 0 at the far plane: xyz IS the direction
825
+ let prev_clip = u.prev_vp * vec4<f32>(dir, 0.0);
826
+ if (prev_clip.w > 0.00001) {
827
+ let prev_ndc = prev_clip.xyz / prev_clip.w;
828
+ prev_uv = vec2<f32>(prev_ndc.x * 0.5 + 0.5, 1.0 - (prev_ndc.y * 0.5 + 0.5));
829
+ } else {
830
+ prev_uv = in.uv;
831
+ }
832
+ } else {
833
+ let prev_clip = u.prev_vp * vec4<f32>(world, 1.0);
834
+ let prev_ndc = prev_clip.xyz / prev_clip.w;
835
+ prev_uv = vec2<f32>(prev_ndc.x * 0.5 + 0.5, 1.0 - (prev_ndc.y * 0.5 + 0.5));
836
+ }
837
+
838
+ var history = current;
839
+ var history_w = current_w;
840
+ if (prev_uv.x >= 0.0 && prev_uv.x <= 1.0 && prev_uv.y >= 0.0 && prev_uv.y <= 1.0) {
841
+ let h_sample = textureSampleLevel(history_tex, history_samp, prev_uv, 0.0);
842
+ history = h_sample.rgb;
843
+ history_w = h_sample.a;
844
+ }
845
+
846
+ // Variance clamp in YCoCg (Karis 2014). Per-channel RGB min/max
847
+ // clamping was producing chromatic sparkle on the stone floor
848
+ // at grazing angles: high-frequency normal-map specular makes
849
+ // each jittered frame's Cg/Co vary significantly, and clamping
850
+ // each channel independently lets history's chroma get pinned
851
+ // to whatever the current frame's specific Cg/Co range was.
852
+ // Clamping only the *luma* axis (Y) preserves chroma stability
853
+ // across frames; the 1σ variance range is a statistical clamp
854
+ // that absorbs single-pixel outliers without collapsing to a
855
+ // hard min/max bound.
856
+ let texel = vec2<f32>(1.0 / f32(textureDimensions(composed_tex).x),
857
+ 1.0 / f32(textureDimensions(composed_tex).y));
858
+ // Neighborhood statistics track the same unjittered position, or the
859
+ // clamp window would wobble against the sample it bounds.
860
+ let center_rgb = textureSampleLevel(composed_tex, composed_samp, src_uv, 0.0).rgb;
861
+ var m1 = rgb_to_ycocg(center_rgb);
862
+ var m2 = m1 * m1;
863
+ let n_samples = 9.0;
864
+ for (var y = -1; y <= 1; y = y + 1) {
865
+ for (var x = -1; x <= 1; x = x + 1) {
866
+ if (x == 0 && y == 0) { continue; }
867
+ let s_uv = src_uv + vec2<f32>(f32(x), f32(y)) * texel;
868
+ let s_rgb = textureSampleLevel(composed_tex, composed_samp, s_uv, 0.0).rgb;
869
+ let s = rgb_to_ycocg(s_rgb);
870
+ m1 = m1 + s;
871
+ m2 = m2 + s * s;
872
+ }
873
+ }
874
+ let mean = m1 / n_samples;
875
+ let variance = max(m2 / n_samples - mean * mean, vec3<f32>(0.0));
876
+ let stddev = sqrt(variance);
877
+
878
+ // Motion-aware γ + alpha. At rest γ=1.25 lets sub-pixel jitter
879
+ // history through for smooth accumulation. Under any camera
880
+ // motion γ collapses fast to 0.25 — forces reprojected
881
+ // history within a quarter-sigma of the neighborhood mean,
882
+ // which is tight enough to reject the 'dark column in
883
+ // history, bright wall in current' case that the wider band
884
+ // let slip. alpha ramps to 0.85 at the same time so remaining
885
+ // history contributes only 15 %.
886
+ let motion_alpha = smoothstep(0.0005, 0.008, vel_len);
887
+ let gamma = mix(1.25, 0.25, motion_alpha);
888
+ let y_min = mean.x - gamma * stddev.x;
889
+ let y_max = mean.x + gamma * stddev.x;
890
+
891
+ let history_ycocg = rgb_to_ycocg(history);
892
+ let history_y_clamped = clamp(history_ycocg.x, y_min, y_max);
893
+ // Chroma is clamped too, but at 3x the luma band (flicker fix).
894
+ // Fully unclamped chroma let stale history colour bleed through on
895
+ // high-contrast edges — green terrain fringing crawling along cloud
896
+ // and canopy silhouettes during camera motion. The loose band keeps
897
+ // the anti-sparkle intent of the luma-only design (a hard
898
+ // per-channel clamp caused chromatic sparkle on grazing stone)
899
+ // while bounding gross cross-object colour bleed.
900
+ let c_gamma = gamma * 3.0;
901
+ let co_clamped = clamp(history_ycocg.y,
902
+ mean.y - c_gamma * stddev.y, mean.y + c_gamma * stddev.y);
903
+ let cg_clamped = clamp(history_ycocg.z,
904
+ mean.z - c_gamma * stddev.z, mean.z + c_gamma * stddev.z);
905
+ let clamped_history = ycocg_to_rgb(vec3<f32>(history_y_clamped, co_clamped, cg_clamped));
906
+
907
+ // Per-pixel disocclusion reject. If the history (already
908
+ // variance-clamped) still sits far from the current
909
+ // neighborhood's center, the reprojection sampled a very
910
+ // different world point and should be dropped. Absolute
911
+ // threshold keyed to stddev so tight-gradient regions
912
+ // reject aggressively; flat regions stay accumulating.
913
+ // NOTE: do not gate this by motion — the luma-only variance clamp
914
+ // deliberately leaves chroma unbounded, and this reject is what
915
+ // flushes chroma-poisoned history on static pixels (weakening it
916
+ // green-tints the whole frame within seconds).
917
+ let history_dist = abs(history_y_clamped - mean.x);
918
+ // The reject's lower edge is MOTION-AWARE (2026-07-16 sparkle hunt).
919
+ // At rest, sub-pixel jitter walks high-frequency detail (pebble bank,
920
+ // grass tips) across the 0.25-sigma edge every few frames, so alpha
921
+ // oscillated between converged history and raw current - measured as
922
+ // 2x MORE temporal flicker with TAA on than off on static surfaces
923
+ // (bank 3.99 vs 1.94 high-pass stddev). Rest pixels now need 0.6 sigma
924
+ // before rejecting; under motion the tight 0.25 returns (the
925
+ // dark-column-in-history case lives there). The chroma-poison flush
926
+ // this reject exists for still fires at rest: poisoned history sits
927
+ // at ~1 sigma+, well past either edge.
928
+ let dis_lo = stddev.x * mix(0.6, 0.25, motion_alpha);
929
+ let disocclusion = smoothstep(dis_lo, stddev.x * 1.0, history_dist);
930
+
931
+ let motion_ramped = mix(u.params.x, 0.85, motion_alpha);
932
+ let alpha = max(motion_ramped, disocclusion);
933
+ let blended = mix(clamped_history, current, alpha);
934
+ let blended_w = mix(history_w, current_w, alpha);
935
+ return vec4<f32>(blended, blended_w);
936
+ }
937
+ ";
938
+
939
+ /// Auto-exposure update shader. Runs at 1×1 viewport → single
940
+ /// fragment. Samples hdr_rt at a 4×4 grid (16 taps), averages
941
+ /// luminance, derives a target exposure via `key / avg_luma`,
942
+ /// smooths toward it from last frame's exposure. One fragment's
943
+ /// worth of work — way cheaper than having every composite
944
+ /// fragment redundantly do the same average.
945
+ pub(in crate::renderer) const EXPOSURE_SHADER_WGSL: &str = "
946
+ struct ExposureParams {
947
+ /// x = target key value (0.18 = photography 18%-gray).
948
+ /// y = smoothing rate (0 = no adapt, 1 = instant).
949
+ /// z = min exposure clamp (prevents pitch-black scenes from
950
+ /// exploding to max brightness).
951
+ /// w = max exposure clamp (prevents sun scenes from crushing
952
+ /// to zero).
953
+ params: vec4<f32>,
954
+ };
955
+
956
+ @group(0) @binding(0) var<uniform> u: ExposureParams;
957
+ @group(0) @binding(1) var hdr_tex: texture_2d<f32>;
958
+ @group(0) @binding(2) var hdr_samp: sampler;
959
+ @group(0) @binding(3) var prev_exposure_tex: texture_2d<f32>;
960
+ @group(0) @binding(4) var prev_exposure_samp: sampler;
961
+
962
+ @vertex
963
+ fn vs_main(@builtin(vertex_index) vid: u32) -> @builtin(position) vec4<f32> {
964
+ let x = f32((vid & 1u) * 4u) - 1.0;
965
+ let y = f32((vid >> 1u) * 4u) - 1.0;
966
+ return vec4<f32>(x, y, 0.0, 1.0);
967
+ }
968
+
969
+ @fragment
970
+ fn fs_main() -> @location(0) vec4<f32> {
971
+ // Histogram-based auto-exposure. 1024-tap (32×32) log-luma
972
+ // sampling into 64 bins, then target the 50th-percentile
973
+ // (median) luma. Much more robust than log-average on scenes
974
+ // with small bright outliers (windows, sun, skylights) — the
975
+ // median ignores outliers while the average gets dragged
976
+ // toward them.
977
+ var bins: array<u32, 64>;
978
+ for (var i = 0u; i < 64u; i = i + 1u) { bins[i] = 0u; }
979
+
980
+ // Histogram log-luma range: 2^-8 (≈0.004) to 2^6 (64). Covers
981
+ // the common HDR exposure range for natural scenes; values
982
+ // outside get clamped into the edge bins so they still count.
983
+ let log_min = -8.0;
984
+ let log_max = 6.0;
985
+ let log_range = log_max - log_min;
986
+
987
+ var total: u32 = 0u;
988
+ let n = 32u;
989
+ for (var y = 0u; y < n; y = y + 1u) {
990
+ for (var x = 0u; x < n; x = x + 1u) {
991
+ let sx = (f32(x) + 0.5) / f32(n);
992
+ let sy = (f32(y) + 0.5) / f32(n);
993
+ let s = textureSample(hdr_tex, hdr_samp, vec2<f32>(sx, sy)).rgb;
994
+ let luma = max(dot(s, vec3<f32>(0.2126, 0.7152, 0.0722)), 1e-4);
995
+ let lg = log2(luma);
996
+ let t = clamp((lg - log_min) / log_range, 0.0, 0.9999);
997
+ let bin = u32(t * 64.0);
998
+ bins[bin] = bins[bin] + 1u;
999
+ total = total + 1u;
1000
+ }
1001
+ }
1002
+
1003
+ // Find the bin whose cumulative-below-count passes 50% of total.
1004
+ let target_count = total / 2u;
1005
+ var accum: u32 = 0u;
1006
+ var median_bin: u32 = 32u;
1007
+ for (var i = 0u; i < 64u; i = i + 1u) {
1008
+ accum = accum + bins[i];
1009
+ if (accum >= target_count) {
1010
+ median_bin = i;
1011
+ break;
1012
+ }
1013
+ }
1014
+ // Interpolate the percentile position WITHIN the bin (flicker fix).
1015
+ // Bin centres quantized the target in whole-bin (~0.22 log2 ≈ 16%)
1016
+ // steps, so histogram noise flipping the median across a bin
1017
+ // boundary stepped the exposure hard enough for TAA to reject its
1018
+ // history — a visible sharp/soft convergence cycle on detailed
1019
+ // surfaces. The cumulative counts turn the bin index into a
1020
+ // continuous position instead.
1021
+ let below = accum - bins[median_bin];
1022
+ var frac = 0.5;
1023
+ if (bins[median_bin] > 0u) {
1024
+ frac = clamp(
1025
+ (f32(target_count) - f32(below)) / f32(bins[median_bin]),
1026
+ 0.0, 1.0);
1027
+ }
1028
+ let median_log = log_min + (f32(median_bin) + frac) / 64.0 * log_range;
1029
+ let median_luma = exp2(median_log);
1030
+
1031
+ let key = u.params.x;
1032
+ let rate = u.params.y;
1033
+ let min_e = u.params.z;
1034
+ let max_e = u.params.w;
1035
+
1036
+ let raw_target = clamp(key / max(median_luma, 0.01), min_e, max_e);
1037
+ let prev = textureSample(prev_exposure_tex, prev_exposure_samp, vec2<f32>(0.5, 0.5));
1038
+ // Anchored deadband (flicker fix): once converged, the exposure must
1039
+ // not chase sub-2% measurement noise (TAA jitter, foliage sway).
1040
+ // .g holds the anchored target; it only re-anchors when the raw
1041
+ // measurement strays more than 2% from it, so a static scene gets a
1042
+ // byte-stable exposure while real lighting changes re-anchor at
1043
+ // once and adapt at the normal rate.
1044
+ var anchor = prev.g;
1045
+ if (anchor < min_e * 0.5 || abs(raw_target - anchor) > anchor * 0.02) {
1046
+ anchor = raw_target;
1047
+ }
1048
+ // Proportional adaptation speed (flicker fix): the measurement is
1049
+ // downstream of TAA, so its residual wiggle produces small target
1050
+ // corrections; ramping them at full rate reads as a visible
1051
+ // brightness sweep that the tonemap + sharpen chain amplifies on
1052
+ // fine texture. Small gaps close glacially (imperceptible per
1053
+ // frame); a real scene change (>25% luminance) adapts at the full
1054
+ // authored rate.
1055
+ let gap = abs(anchor - prev.r) / max(prev.r, 1e-3);
1056
+ let rate_scale = smoothstep(0.0, 0.25, gap);
1057
+ // First frame: prev is 0; snap to target instead of crawling up.
1058
+ var smoothed = mix(prev.r, anchor, rate * rate_scale);
1059
+ if (prev.r < min_e * 0.5) {
1060
+ smoothed = anchor;
1061
+ }
1062
+ return vec4<f32>(smoothed, anchor, 0.0, 1.0);
1063
+ }
1064
+ ";
1065
+
1066
+ /// Composite + tonemap fragment shader. Single fullscreen triangle
1067
+ /// reads hdr_rt and writes ACES-tonemapped linear-RGB. Hardware
1068
+ /// performs the linear→sRGB encode on write because the surface
1069
+ /// format is sRGB.
1070
+ pub(in crate::renderer) const COMPOSITE_SHADER_WGSL: &str = "
1071
+ struct CompositeParams {
1072
+ /// x = tonemap mode (0 = ACES, 1 = AgX)
1073
+ /// y = auto-exposure enabled (0 = off, uses manual x)
1074
+ /// z = manual exposure multiplier (used when auto is off)
1075
+ /// w = auto-exposure target key value (0.18 = 18% gray photo standard)
1076
+ params: vec4<f32>,
1077
+ /// Filmic-look knobs — all default to 0 (effect off).
1078
+ /// x = chromatic aberration strength (0..~0.01 radial UV offset)
1079
+ /// y = vignette strength (0..1, darkens corners)
1080
+ /// z = vignette softness (0..1, smaller = harder edge)
1081
+ /// w = film grain strength (0..~0.1 amplitude added to luma)
1082
+ filmic: vec4<f32>,
1083
+ /// x = grain seed (frame index, randomizes the noise per frame);
1084
+ /// y = sharpen strength (0 = off, ~0.25 subtle, ~0.5 punchy);
1085
+ /// zw padding.
1086
+ misc: vec4<f32>,
1087
+ };
1088
+
1089
+ @group(0) @binding(0) var hdr_tex: texture_2d<f32>;
1090
+ @group(0) @binding(1) var hdr_samp: sampler;
1091
+ @group(0) @binding(2) var<uniform> u: CompositeParams;
1092
+ @group(0) @binding(3) var exposure_tex: texture_2d<f32>;
1093
+ @group(0) @binding(4) var exposure_samp: sampler;
1094
+ @group(0) @binding(5) var ssao_tex: texture_2d<f32>;
1095
+ @group(0) @binding(6) var ssao_samp: sampler;
1096
+
1097
+ struct VsOut {
1098
+ @builtin(position) clip_pos: vec4<f32>,
1099
+ @location(0) uv: vec2<f32>,
1100
+ };
1101
+
1102
+ @vertex
1103
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
1104
+ let x = f32((vid & 1u) * 4u) - 1.0;
1105
+ let y = f32((vid >> 1u) * 4u) - 1.0;
1106
+ var out: VsOut;
1107
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
1108
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
1109
+ return out;
1110
+ }
1111
+
1112
+ fn aces_tone(c: vec3<f32>) -> vec3<f32> {
1113
+ let a = 2.51;
1114
+ let b = 0.03;
1115
+ let cc = 2.43;
1116
+ let d = 0.59;
1117
+ let e = 0.14;
1118
+ return clamp((c * (c * a + b)) / (c * (c * cc + d) + e), vec3<f32>(0.0), vec3<f32>(1.0));
1119
+ }
1120
+
1121
+ // --- AgX tonemap (Blender/Filament reference) ---
1122
+ // Better hue preservation than ACES in saturated regions — reds
1123
+ // stay red instead of shifting toward orange, blues stay blue
1124
+ // instead of shifting toward cyan. Same sigmoid shape overall,
1125
+ // so the overall contrast is similar.
1126
+
1127
+ fn agx_default_contrast_approx(x: vec3<f32>) -> vec3<f32> {
1128
+ let x2 = x * x;
1129
+ let x4 = x2 * x2;
1130
+ return vec3<f32>(15.5) * x4 * x2
1131
+ - vec3<f32>(40.14) * x4 * x
1132
+ + vec3<f32>(31.96) * x4
1133
+ - vec3<f32>(6.868) * x2 * x
1134
+ + vec3<f32>(0.4298) * x2
1135
+ + vec3<f32>(0.1191) * x
1136
+ - vec3<f32>(0.00232);
1137
+ }
1138
+
1139
+ fn agx_tone(val_in: vec3<f32>) -> vec3<f32> {
1140
+ // AgX input transform — compresses the input color gamut.
1141
+ let agx_mat = mat3x3<f32>(
1142
+ vec3<f32>(0.842479062253094, 0.0423282422610123, 0.0423756549057051),
1143
+ vec3<f32>(0.0784335999999992, 0.878468636469772, 0.0784336),
1144
+ vec3<f32>(0.0792237451477643, 0.0791661274605434, 0.879142973793104),
1145
+ );
1146
+ // Log2-space normalization range. Anything outside gets clamped
1147
+ // — the sigmoid maps this window to [0, 1].
1148
+ let min_ev = -12.47393;
1149
+ let max_ev = 4.026069;
1150
+
1151
+ var val = agx_mat * val_in;
1152
+ // Log2 encode, clamp to range, normalize to [0, 1].
1153
+ val = max(val, vec3<f32>(1e-10));
1154
+ val = clamp(log2(val), vec3<f32>(min_ev), vec3<f32>(max_ev));
1155
+ val = (val - vec3<f32>(min_ev)) / (max_ev - min_ev);
1156
+
1157
+ // Sigmoid contrast curve.
1158
+ val = agx_default_contrast_approx(val);
1159
+ return val;
1160
+ }
1161
+
1162
+ fn agx_eotf(val_in: vec3<f32>) -> vec3<f32> {
1163
+ // AgX inverse input transform — re-expands back to target
1164
+ // display gamut. The surface is sRGB-format so hardware
1165
+ // applies the sRGB EOTF on write; we output linear here.
1166
+ let agx_mat_inv = mat3x3<f32>(
1167
+ vec3<f32>( 1.19687900512017, -0.0528968517574562, -0.0529716355144438),
1168
+ vec3<f32>(-0.0980208811401368, 1.15190312990417, -0.0980434501171241),
1169
+ vec3<f32>(-0.0990297440797205, -0.0989611768448433, 1.15107367264116),
1170
+ );
1171
+ return agx_mat_inv * val_in;
1172
+ }
1173
+
1174
+ // Tonemap — branches on u.params.x (0 = ACES, 1 = AgX). Extracted
1175
+ // so the sharpen pass can tonemap neighbour HDR samples through the
1176
+ // same path as the center pixel.
1177
+ fn tonemap_select(hdr: vec3<f32>) -> vec3<f32> {
1178
+ if (u.params.x < 0.5) {
1179
+ return aces_tone(hdr);
1180
+ }
1181
+ // Full Filament AgX pipeline:
1182
+ // agx_tone (inset + log2 + sigmoid)
1183
+ // agx_eotf (outset = inverse inset)
1184
+ // pow(2.2) (display-encoded → linear, required for sRGB surface)
1185
+ //
1186
+ // Filament's polynomial-approximation AgX (what we use) systematically
1187
+ // under-saturates vs Blender/Cycles' LUT-based AgX. Apply the 'Punchy'
1188
+ // look — saturation 1.4, slight contrast — as a post step to bring
1189
+ // colours closer to the Cycles ground-truth reference. Matches
1190
+ // Filament::ToneMapper::AGX_PUNCHY.
1191
+ let hdr_safe = max(hdr, vec3<f32>(0.0));
1192
+ let agx_display = clamp(agx_eotf(agx_tone(hdr_safe)), vec3<f32>(0.0), vec3<f32>(1.0));
1193
+ let linear = pow(agx_display, vec3<f32>(2.2));
1194
+ // Punchy look: post-tonemap saturation + contrast in display space.
1195
+ // slope/offset/power are per-channel lift/gamma/gain; saturation
1196
+ // multiplies chroma around the pixel's luma.
1197
+ let agx_punchy = agx_look_punchy(linear);
1198
+ return agx_punchy;
1199
+ }
1200
+
1201
+ fn agx_look_punchy(val: vec3<f32>) -> vec3<f32> {
1202
+ // Per-channel slope (gain): subtle warm white-balance toward the
1203
+ // 'golden hour' feel of Cycles-AgX renders — +3% red, -2% blue.
1204
+ // The Bistro outdoor.hdr leans cool; without this, the shadows
1205
+ // and IBL fill pull the overall image toward blue-grey vs.
1206
+ // Cycles's warmer midtone neutral.
1207
+ let slope = vec3<f32>(1.03, 1.00, 0.98);
1208
+ let offset = vec3<f32>(0.0);
1209
+ // A whisker above neutral — Cycles-AgX with its LUT-based DRT
1210
+ // preserves enough chroma on its own that our polynomial fit only
1211
+ // needs a gentle post-boost to catch up.
1212
+ let power = vec3<f32>(1.1);
1213
+ let saturation = 1.1;
1214
+ // ASC-CDL-ish: (val * slope + offset) ^ power
1215
+ let toned = pow(max(val * slope + offset, vec3<f32>(0.0)), power);
1216
+ let luma = dot(toned, vec3<f32>(0.2126, 0.7152, 0.0722));
1217
+ return vec3<f32>(luma) + saturation * (toned - vec3<f32>(luma));
1218
+ }
1219
+
1220
+ // Hash-based pseudo-random in [0, 1). Cheap noise function for grain;
1221
+ // not great for cryptography or stratified sampling, but visually
1222
+ // indistinguishable from white noise at film-grain strengths.
1223
+ fn hash21(p: vec2<f32>) -> f32 {
1224
+ var p3 = fract(vec3<f32>(p.xyx) * 0.1031);
1225
+ p3 = p3 + dot(p3, p3.yzx + 33.33);
1226
+ return fract((p3.x + p3.y) * p3.z);
1227
+ }
1228
+
1229
+ @fragment
1230
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
1231
+ // Composite samples the TAA-blended HDR. The TAA pass has
1232
+ // already combined HDR + SSAO + bloom + SSR + fog into one
1233
+ // linear-HDR value, so all that's left is exposure, tonemap,
1234
+ // and the optional filmic-look layer (CA / vignette / grain).
1235
+ var sample_uv = in.uv;
1236
+
1237
+ // Sample the composed HDR at the centre pixel. Chromatic
1238
+ // aberration used to sample here at 3 offsets and take R/G/B
1239
+ // from different UVs, but doing that on pre-tonemap HDR
1240
+ // (values 1-20) turns any sub-pixel specular highlight into
1241
+ // a magenta/green speckle: the R tap lands on a bright pixel,
1242
+ // the B tap on a dark one, and the colour ratio survives
1243
+ // tonemap as visible noise all over the frame. CA was moved
1244
+ // after tonemap below, where the values are bounded to [0,1]
1245
+ // and the offset reads a colour difference of a few percent
1246
+ // instead of a 10× ratio.
1247
+ let centre_sample = textureSample(hdr_tex, hdr_samp, sample_uv);
1248
+ let indirect_weight = centre_sample.a;
1249
+
1250
+ // Last-chance NaN/Inf scrub before tonemap. The main HDR pass
1251
+ // already self-compares its output, but compose mixes SSR + SSGI
1252
+ // + bloom + fog, any of which can introduce a non-finite value
1253
+ // (degenerate view rays, env UV pole, multi-scatter fresnel
1254
+ // divide). With TAA off, no neighborhood clamp exists upstream
1255
+ // to catch it — and tonemap(NaN) on Metal produces pink.
1256
+ let hdr_raw = select(vec3<f32>(0.0), centre_sample.rgb, centre_sample.rgb == centre_sample.rgb);
1257
+
1258
+ // SSAO RT packs AO in R and contact shadow in G — multiplying
1259
+ // both gives combined darkening. The AO channel is bilaterally
1260
+ // blurred; G is the raw pixel-accurate contact result.
1261
+ //
1262
+ // Apply AO only in proportion to how INDIRECT this pixel's light
1263
+ // is. indirect_weight = 1 means fully shadowed (indirect lighting
1264
+ // only — full AO applies), weight = 0 means fully sunlit (direct
1265
+ // lighting dominates — AO shouldn't darken it). That's physically
1266
+ // correct: AO models the fact that nearby geometry occludes
1267
+ // ambient/bounce light, but it has nothing to say about a direct
1268
+ // ray from the sun, which the shadow map already handles. Sky
1269
+ // pixels get indirect_weight = 0 via the scene_compose pass (sky
1270
+ // albedo = 0) so AO also leaves them alone.
1271
+ let ao_pair = textureSample(ssao_tex, ssao_samp, sample_uv).rg;
1272
+ let ao_combined = ao_pair.r * ao_pair.g;
1273
+ let ao_weighted = mix(1.0, ao_combined, indirect_weight);
1274
+ let hdr_ao = hdr_raw * ao_weighted;
1275
+
1276
+ // Exposure. Two modes:
1277
+ // auto off → manual exposure multiplier (u.params.z).
1278
+ // auto on → read the smoothed exposure value from a 1×1
1279
+ // texture populated by the exposure update pass.
1280
+ var exposure: f32;
1281
+ if (u.params.y < 0.5) {
1282
+ exposure = u.params.z;
1283
+ } else {
1284
+ exposure = textureSample(exposure_tex, exposure_samp, vec2<f32>(0.5, 0.5)).r;
1285
+ }
1286
+ let hdr = hdr_ao * exposure;
1287
+
1288
+ // Branch between ACES and AgX via the uniform. Costs one
1289
+ // compare per fragment; the dead branch gets DCE'd per-draw
1290
+ // since the uniform is constant across the frame.
1291
+ var ldr = tonemap_select(hdr);
1292
+
1293
+ // --- Chromatic aberration (UE5 formula, post-tonemap) ---
1294
+ // Port of the CA block in UE5's PostProcessTonemap.usf lines
1295
+ // 320-342. Three pieces we take directly from Epic's shader:
1296
+ //
1297
+ // 1. LensUV-space (-1..+1) per-axis formula with a StartOffset
1298
+ // dead-zone around the centre: shift = sign(lens) *
1299
+ // saturate(|lens| - start) * scale. Inside the dead-zone
1300
+ // the centre of the frame stays perfectly sharp; only the
1301
+ // edges pick up the fringe, matching the way a real
1302
+ // photographic lens disperses light at the periphery.
1303
+ //
1304
+ // 2. Separate R and B scales for wavelength dispersion.
1305
+ // Red refracts less than blue in a real lens, so the B
1306
+ // shift is ~1.5× the R shift (matches the typical default
1307
+ // that ships in UE5's post-process volume).
1308
+ //
1309
+ // 3. Green is sampled at the centre UV (no shift). Only R and
1310
+ // B fringe outward.
1311
+ //
1312
+ // We diverge from UE5 in one spot: UE5 samples HDR pre-tonemap
1313
+ // and relies on TAA's neighborhood clamp to bound the HDR
1314
+ // ratios. With TAA optional here, we run the offset samples
1315
+ // through the same AO + exposure + tonemap path and compose R
1316
+ // and B from the LDR results — so a bright specular firefly
1317
+ // can't ride the HDR ratio and survive tonemap as a magenta /
1318
+ // green speck across the whole frame.
1319
+ let ca_strength = u.filmic.x;
1320
+ if (ca_strength > 0.0) {
1321
+ let ca_r_scale = ca_strength;
1322
+ let ca_b_scale = ca_strength * 1.5;
1323
+ let start_offset = 0.25;
1324
+
1325
+ let lens_uv = sample_uv * 2.0 - 1.0;
1326
+ let beyond = max(abs(lens_uv) - vec2<f32>(start_offset), vec2<f32>(0.0));
1327
+ let sign_lens = sign(lens_uv);
1328
+ let uv_r_lens = lens_uv - sign_lens * beyond * ca_r_scale;
1329
+ let uv_b_lens = lens_uv - sign_lens * beyond * ca_b_scale;
1330
+ let uv_r = uv_r_lens * 0.5 + 0.5;
1331
+ let uv_b = uv_b_lens * 0.5 + 0.5;
1332
+
1333
+ let s_r = textureSample(hdr_tex, hdr_samp, uv_r);
1334
+ let s_b = textureSample(hdr_tex, hdr_samp, uv_b);
1335
+ let clean_r = select(vec3<f32>(0.0), s_r.rgb, s_r.rgb == s_r.rgb);
1336
+ let clean_b = select(vec3<f32>(0.0), s_b.rgb, s_b.rgb == s_b.rgb);
1337
+ let ao_r_pair = textureSample(ssao_tex, ssao_samp, uv_r).rg;
1338
+ let ao_b_pair = textureSample(ssao_tex, ssao_samp, uv_b).rg;
1339
+ let ao_r = mix(1.0, ao_r_pair.r * ao_r_pair.g, s_r.a);
1340
+ let ao_b = mix(1.0, ao_b_pair.r * ao_b_pair.g, s_b.a);
1341
+ let ldr_r_full = tonemap_select(clean_r * ao_r * exposure);
1342
+ let ldr_b_full = tonemap_select(clean_b * ao_b * exposure);
1343
+ ldr = vec3<f32>(ldr_r_full.r, ldr.g, ldr_b_full.b);
1344
+ }
1345
+
1346
+ // --- Sharpen (post-tonemap unsharp mask) ---
1347
+ // Subtle lens-like crispening. Samples 4 neighbour HDR values,
1348
+ // averages them in HDR, then runs ONE tonemap on the average and
1349
+ // adds the (centre - avg) difference back scaled by
1350
+ // `sharpen_strength`. Operating in LDR post-tonemap avoids the
1351
+ // classic problem of HDR sharpen blowing out highlights (the
1352
+ // unsharp of a bright pixel against a dark one gets amplified
1353
+ // into an ugly rim). tonemap(avg) instead of avg(tonemap) is a
1354
+ // deliberate approximation: for a 1-px cross kernel the two agree
1355
+ // to well under a percent except on hard HDR edges — where the
1356
+ // clamp bounds the difference anyway — and it turns 4 extra
1357
+ // tonemap evaluations per pixel into 1. At 4K output the exact
1358
+ // version measurably dominated the whole composite pass.
1359
+ let sharpen_strength = u.misc.y;
1360
+ if (sharpen_strength > 0.0) {
1361
+ let dims = vec2<f32>(textureDimensions(hdr_tex));
1362
+ let t = vec2<f32>(1.0 / dims.x, 1.0 / dims.y);
1363
+ let ox = vec2<f32>(t.x, 0.0);
1364
+ let oy = vec2<f32>(0.0, t.y);
1365
+ let h_r = textureSample(hdr_tex, hdr_samp, sample_uv + ox).rgb;
1366
+ let h_l = textureSample(hdr_tex, hdr_samp, sample_uv - ox).rgb;
1367
+ let h_d = textureSample(hdr_tex, hdr_samp, sample_uv + oy).rgb;
1368
+ let h_u = textureSample(hdr_tex, hdr_samp, sample_uv - oy).rgb;
1369
+ let h_avg = (h_r + h_l + h_d + h_u) * 0.25 * ao_weighted * exposure;
1370
+ let avg = tonemap_select(h_avg);
1371
+ // Flicker fix: cap the unsharp detail term. On a high-contrast
1372
+ // silhouette (building roofline vs sky) detail is huge, and at
1373
+ // half-res TSR the reconstructed edge wobbles a sub-pixel every
1374
+ // frame — a strong unsharp turns that wobble into a crawling
1375
+ // bright/dark line (the reported gray lines on the building).
1376
+ // Fine texture detail has small local contrast and passes
1377
+ // through untouched; only the extreme edge overshoot is bounded,
1378
+ // which also removes the silhouette halo the 0.8 default caused.
1379
+ let detail = clamp(ldr - avg, vec3<f32>(-0.12), vec3<f32>(0.12));
1380
+ ldr = clamp(ldr + detail * sharpen_strength, vec3<f32>(0.0), vec3<f32>(1.0));
1381
+ }
1382
+
1383
+ // --- Vignette (post-tonemap) ---
1384
+ // Smooth radial darkening. Applied after tonemap so it stays
1385
+ // perceptually uniform across exposures (otherwise bright
1386
+ // scenes wash out the vignette).
1387
+ let vig_strength = u.filmic.y;
1388
+ if (vig_strength > 0.0) {
1389
+ let vig_softness = max(u.filmic.z, 0.001);
1390
+ let dist = length(in.uv - vec2<f32>(0.5, 0.5));
1391
+ // smoothstep gives a natural falloff; remap so strength=1
1392
+ // fully blackens the corner and softness controls width.
1393
+ let edge = smoothstep(0.5 - vig_softness, 0.75, dist);
1394
+ ldr = ldr * (1.0 - edge * vig_strength);
1395
+ }
1396
+
1397
+ // --- Film grain (post-tonemap) ---
1398
+ // Per-pixel noise added to luma. Animated by frame seed in
1399
+ // misc.x so grain crawls naturally; if seed stays fixed (e.g.
1400
+ // headless screenshots) the grain freezes.
1401
+ let grain_strength = u.filmic.w;
1402
+ if (grain_strength > 0.0) {
1403
+ let seed = u.misc.x;
1404
+ let n = hash21(in.uv * 1024.0 + vec2<f32>(seed, seed * 1.7)) - 0.5;
1405
+ ldr = ldr + vec3<f32>(n * grain_strength);
1406
+ }
1407
+
1408
+ return vec4<f32>(ldr, 1.0);
1409
+ }
1410
+ ";
1411
+
1412
+ // -----------------------------------------------------------------
1413
+ // Upscale pass — render-res → full-surface. Engages only when
1414
+ // `render_scale < 1.0 && !taa_enabled` (when TAA is on the TAA pass
1415
+ // does its own Catmull-Rom reconstruction). Mode 0 = bilinear (cheap,
1416
+ // soft), mode 1 = Catmull-Rom 5-tap (sharper edge reconstruction,
1417
+ // same kernel as the TAA pass).
1418
+ // -----------------------------------------------------------------
1419
+ pub(in crate::renderer) const UPSCALE_SHADER_WGSL: &str = "
1420
+ struct UpscaleParams {
1421
+ // x = mode (0 = bilinear, 1 = catmull-rom), yzw padding.
1422
+ params: vec4<f32>,
1423
+ };
1424
+
1425
+ @group(0) @binding(0) var<uniform> u: UpscaleParams;
1426
+ @group(0) @binding(1) var composed_tex: texture_2d<f32>;
1427
+ @group(0) @binding(2) var composed_samp: sampler;
1428
+
1429
+ struct VsOut {
1430
+ @builtin(position) clip_pos: vec4<f32>,
1431
+ @location(0) uv: vec2<f32>,
1432
+ };
1433
+
1434
+ @vertex
1435
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
1436
+ let x = f32((vid & 1u) * 4u) - 1.0;
1437
+ let y = f32((vid >> 1u) * 4u) - 1.0;
1438
+ var out: VsOut;
1439
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
1440
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
1441
+ return out;
1442
+ }
1443
+
1444
+ // 5-tap Catmull-Rom (Karis formulation) — same kernel as the TAA
1445
+ // shader's upsample. Costs 5 bilinear fetches vs 1; reconstructs a
1446
+ // cubic-Hermite curve through 4 source taps which preserves edges
1447
+ // where naive bilinear goes mushy.
1448
+ fn sample_catmull_rom(uv: vec2<f32>) -> vec4<f32> {
1449
+ let tex_size = vec2<f32>(textureDimensions(composed_tex));
1450
+ let inv_size = 1.0 / tex_size;
1451
+ let sample_pos = uv * tex_size;
1452
+ let tex_pos1 = floor(sample_pos - 0.5) + 0.5;
1453
+ let f = sample_pos - tex_pos1;
1454
+ let w0 = f * (-0.5 + f * (1.0 - 0.5 * f));
1455
+ let w1 = 1.0 + f * f * (-2.5 + 1.5 * f);
1456
+ let w2 = f * (0.5 + f * (2.0 - 1.5 * f));
1457
+ let w3 = f * f * (-0.5 + 0.5 * f);
1458
+ let w12 = w1 + w2;
1459
+ let offset12 = w2 / w12;
1460
+ let tp0 = (tex_pos1 - 1.0) * inv_size;
1461
+ let tp3 = (tex_pos1 + 2.0) * inv_size;
1462
+ let tp12 = (tex_pos1 + offset12) * inv_size;
1463
+ var result = vec4<f32>(0.0);
1464
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp12.x, tp0.y), 0.0) * w12.x * w0.y;
1465
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp0.x, tp12.y), 0.0) * w0.x * w12.y;
1466
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp12.x, tp12.y), 0.0) * w12.x * w12.y;
1467
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp3.x, tp12.y), 0.0) * w3.x * w12.y;
1468
+ result += textureSampleLevel(composed_tex, composed_samp, vec2<f32>(tp12.x, tp3.y), 0.0) * w12.x * w3.y;
1469
+ return max(result, vec4<f32>(0.0));
1470
+ }
1471
+
1472
+ @fragment
1473
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
1474
+ let mode = u32(u.params.x);
1475
+ if (mode == 1u) {
1476
+ return sample_catmull_rom(in.uv);
1477
+ }
1478
+ return textureSample(composed_tex, composed_samp, in.uv);
1479
+ }
1480
+ ";
1481
+
1482
+ // -----------------------------------------------------------------
1483
+ // Contrast-adaptive sharpen — a simplified RCAS (FidelityFX). 5-tap
1484
+ // cross kernel, negative-lobe FIR with lobe amplitude adapted to
1485
+ // per-pixel luma headroom so flat areas don't amplify noise. Runs
1486
+ // before composite on whichever texture feeds composite today
1487
+ // (sss/mb/dof/taa/upscale/composed). Gated on `strength > 0`;
1488
+ // default 0 so the pass is a no-op unless the user opts in.
1489
+ // -----------------------------------------------------------------
1490
+ pub(in crate::renderer) const RCAS_SHADER_WGSL: &str = "
1491
+ struct RcasParams {
1492
+ // x = strength (0 = off, 0.3 = subtle, 0.6 = punchy, 1.0 = max).
1493
+ // yzw padding.
1494
+ params: vec4<f32>,
1495
+ };
1496
+
1497
+ @group(0) @binding(0) var<uniform> u: RcasParams;
1498
+ @group(0) @binding(1) var input_tex: texture_2d<f32>;
1499
+ @group(0) @binding(2) var input_samp: sampler;
1500
+
1501
+ struct VsOut {
1502
+ @builtin(position) clip_pos: vec4<f32>,
1503
+ @location(0) uv: vec2<f32>,
1504
+ };
1505
+
1506
+ @vertex
1507
+ fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
1508
+ let x = f32((vid & 1u) * 4u) - 1.0;
1509
+ let y = f32((vid >> 1u) * 4u) - 1.0;
1510
+ var out: VsOut;
1511
+ out.clip_pos = vec4<f32>(x, y, 0.0, 1.0);
1512
+ out.uv = vec2<f32>((x + 1.0) * 0.5, (1.0 - y) * 0.5);
1513
+ return out;
1514
+ }
1515
+
1516
+ @fragment
1517
+ fn fs_main(in: VsOut) -> @location(0) vec4<f32> {
1518
+ let strength = u.params.x;
1519
+ let center = textureSample(input_tex, input_samp, in.uv);
1520
+ if (strength <= 0.0) {
1521
+ return center;
1522
+ }
1523
+ let tex_size = vec2<f32>(textureDimensions(input_tex));
1524
+ let px = 1.0 / tex_size;
1525
+
1526
+ let c = center.rgb;
1527
+ let n = textureSample(input_tex, input_samp, in.uv + vec2<f32>( 0.0, -px.y)).rgb;
1528
+ let s = textureSample(input_tex, input_samp, in.uv + vec2<f32>( 0.0, px.y)).rgb;
1529
+ let w = textureSample(input_tex, input_samp, in.uv + vec2<f32>(-px.x, 0.0)).rgb;
1530
+ let e = textureSample(input_tex, input_samp, in.uv + vec2<f32>( px.x, 0.0)).rgb;
1531
+
1532
+ // Luma-based local min/max for contrast adaptation. Rec. 709.
1533
+ let lw = vec3<f32>(0.2126, 0.7152, 0.0722);
1534
+ let lc = dot(c, lw);
1535
+ let ln = dot(n, lw);
1536
+ let ls = dot(s, lw);
1537
+ let lwl = dot(w, lw);
1538
+ let le = dot(e, lw);
1539
+ let lmin = min(min(min(ln, ls), min(lwl, le)), lc);
1540
+ let lmax = max(max(max(ln, ls), max(lwl, le)), lc);
1541
+
1542
+ // Headroom — how much room is there before we clip at 0 or the
1543
+ // local max? Small in flat areas, large at edges. This is the
1544
+ // 'Robust' part of RCAS: sharpen only where it helps.
1545
+ let headroom = clamp(lmin / max(lmax, 1e-4), 0.0, 1.0);
1546
+
1547
+ // Lobe amplitude — bigger at edges (low headroom), smaller in
1548
+ // flat areas (high headroom). 0.125 cap keeps the kernel stable.
1549
+ let lobe = 0.125 * strength * (1.0 - headroom);
1550
+
1551
+ // Negative-lobe FIR: center*(1+4*lobe) - lobe*(n+s+w+e).
1552
+ // Coefficients sum to 1 → DC preserved.
1553
+ let sharpened = c * (1.0 + 4.0 * lobe) - lobe * (n + s + w + e);
1554
+
1555
+ return vec4<f32>(max(sharpened, vec3<f32>(0.0)), center.a);
1556
+ }
1557
+ ";
1558
+