@bornengine/engine 0.4.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +231 -0
- package/native/android/Cargo.lock +1848 -0
- package/native/android/Cargo.toml +24 -0
- package/native/android/src/lib.rs +702 -0
- package/native/ios/Cargo.lock +1690 -0
- package/native/ios/Cargo.toml +32 -0
- package/native/ios/src/lib.rs +1267 -0
- package/native/linux/Cargo.lock +3279 -0
- package/native/linux/Cargo.toml +29 -0
- package/native/linux/src/lib.rs +1331 -0
- package/native/macos/Cargo.lock +3310 -0
- package/native/macos/Cargo.toml +46 -0
- package/native/macos/src/lib.rs +1302 -0
- package/native/shared/Cargo.lock +1899 -0
- package/native/shared/Cargo.toml +62 -0
- package/native/shared/assets/default_font.ttf +0 -0
- package/native/shared/build.rs +270 -0
- package/native/shared/shaders/common/clouds.wgsl +122 -0
- package/native/shared/shaders/common/fog.wgsl +16 -0
- package/native/shared/shaders/common/foliage_wind.wgsl +98 -0
- package/native/shared/shaders/common/imposter.wgsl +112 -0
- package/native/shared/shaders/common/pbr.wgsl +186 -0
- package/native/shared/shaders/common/shadows.wgsl +186 -0
- package/native/shared/shaders/common/sky.wgsl +8 -0
- package/native/shared/shaders/common/tonemap.wgsl +25 -0
- package/native/shared/shaders/impulse_field.wgsl +57 -0
- package/native/shared/shaders/material_abi.wgsl +383 -0
- package/native/shared/shaders/materials/test_minimal.wgsl +42 -0
- package/native/shared/src/anim_mixer.rs +61 -0
- package/native/shared/src/attach.rs +263 -0
- package/native/shared/src/audio/decode.rs +123 -0
- package/native/shared/src/audio/mod.rs +863 -0
- package/native/shared/src/audio/render.rs +892 -0
- package/native/shared/src/audio/spsc.rs +156 -0
- package/native/shared/src/audio/stream.rs +226 -0
- package/native/shared/src/custom_shaders.rs +104 -0
- package/native/shared/src/decals.rs +245 -0
- package/native/shared/src/drs.rs +211 -0
- package/native/shared/src/engine.rs +261 -0
- package/native/shared/src/ffi.rs +116 -0
- package/native/shared/src/ffi_core/assets.rs +388 -0
- package/native/shared/src/ffi_core/audio_ffi.rs +184 -0
- package/native/shared/src/ffi_core/draw.rs +334 -0
- package/native/shared/src/ffi_core/game_loop.rs +577 -0
- package/native/shared/src/ffi_core/input.rs +234 -0
- package/native/shared/src/ffi_core/mod.rs +127 -0
- package/native/shared/src/ffi_core/models.rs +1154 -0
- package/native/shared/src/ffi_core/ragdoll_ffi.rs +261 -0
- package/native/shared/src/ffi_core/scene.rs +626 -0
- package/native/shared/src/ffi_core/vfx.rs +212 -0
- package/native/shared/src/ffi_core/visual.rs +691 -0
- package/native/shared/src/frame_callbacks.rs +122 -0
- package/native/shared/src/geometry.rs +236 -0
- package/native/shared/src/handles.rs +182 -0
- package/native/shared/src/input.rs +448 -0
- package/native/shared/src/jolt_sys.rs +822 -0
- package/native/shared/src/lib.rs +55 -0
- package/native/shared/src/models.rs +1093 -0
- package/native/shared/src/models_gltf.rs +1280 -0
- package/native/shared/src/particles.rs +391 -0
- package/native/shared/src/physics_jolt.rs +1908 -0
- package/native/shared/src/picking.rs +298 -0
- package/native/shared/src/postfx.rs +345 -0
- package/native/shared/src/profiler.rs +492 -0
- package/native/shared/src/ragdoll.rs +474 -0
- package/native/shared/src/renderer/atmosphere_lut.rs +573 -0
- package/native/shared/src/renderer/brdf_lut.rs +154 -0
- package/native/shared/src/renderer/draw2d.rs +143 -0
- package/native/shared/src/renderer/formats.rs +822 -0
- package/native/shared/src/renderer/froxel.rs +421 -0
- package/native/shared/src/renderer/gi_bake.rs +653 -0
- package/native/shared/src/renderer/graph.rs +462 -0
- package/native/shared/src/renderer/hiz.rs +269 -0
- package/native/shared/src/renderer/hot_reload.rs +390 -0
- package/native/shared/src/renderer/impulse_field.rs +456 -0
- package/native/shared/src/renderer/lighting.rs +154 -0
- package/native/shared/src/renderer/material_instancing.rs +171 -0
- package/native/shared/src/renderer/material_pipeline.rs +700 -0
- package/native/shared/src/renderer/material_system.rs +1996 -0
- package/native/shared/src/renderer/material_system_tests.rs +601 -0
- package/native/shared/src/renderer/material_system_wasm.rs +41 -0
- package/native/shared/src/renderer/mod.rs +12556 -0
- package/native/shared/src/renderer/model_draw.rs +641 -0
- package/native/shared/src/renderer/occlusion.rs +429 -0
- package/native/shared/src/renderer/planar_pass.rs +593 -0
- package/native/shared/src/renderer/planar_reflection.rs +499 -0
- package/native/shared/src/renderer/post_pass.rs +249 -0
- package/native/shared/src/renderer/postfx_chain.rs +728 -0
- package/native/shared/src/renderer/pt_pass.rs +577 -0
- package/native/shared/src/renderer/scene_pass.rs +607 -0
- package/native/shared/src/renderer/shader_include.rs +205 -0
- package/native/shared/src/renderer/shader_library.rs +135 -0
- package/native/shared/src/renderer/shaders/ao.rs +570 -0
- package/native/shared/src/renderer/shaders/core.rs +1243 -0
- package/native/shared/src/renderer/shaders/env.rs +907 -0
- package/native/shared/src/renderer/shaders/gi.rs +810 -0
- package/native/shared/src/renderer/shaders/mod.rs +19 -0
- package/native/shared/src/renderer/shaders/post.rs +1558 -0
- package/native/shared/src/renderer/shaders/pt.rs +1859 -0
- package/native/shared/src/renderer/shaders/ssgi.rs +1586 -0
- package/native/shared/src/renderer/shadow_pass.rs +731 -0
- package/native/shared/src/renderer/ssgi_pass.rs +392 -0
- package/native/shared/src/renderer/ssr_pass.rs +188 -0
- package/native/shared/src/renderer/texture_store.rs +473 -0
- package/native/shared/src/renderer/transient.rs +591 -0
- package/native/shared/src/renderer/types.rs +941 -0
- package/native/shared/src/renderer/util.rs +152 -0
- package/native/shared/src/scene.rs +1362 -0
- package/native/shared/src/sdf_cache.rs +274 -0
- package/native/shared/src/shadows.rs +1036 -0
- package/native/shared/src/staging.rs +102 -0
- package/native/shared/src/string_header.rs +266 -0
- package/native/shared/src/text_renderer.rs +502 -0
- package/native/shared/src/textures.rs +197 -0
- package/native/tvos/Cargo.lock +1693 -0
- package/native/tvos/Cargo.toml +36 -0
- package/native/tvos/metal-patched/Cargo.toml +178 -0
- package/native/tvos/metal-patched/LICENSE-APACHE +201 -0
- package/native/tvos/metal-patched/LICENSE-MIT +25 -0
- package/native/tvos/metal-patched/src/acceleration_structure.rs +667 -0
- package/native/tvos/metal-patched/src/acceleration_structure_pass.rs +108 -0
- package/native/tvos/metal-patched/src/argument.rs +366 -0
- package/native/tvos/metal-patched/src/blitpass.rs +102 -0
- package/native/tvos/metal-patched/src/buffer.rs +71 -0
- package/native/tvos/metal-patched/src/capturedescriptor.rs +76 -0
- package/native/tvos/metal-patched/src/capturemanager.rs +113 -0
- package/native/tvos/metal-patched/src/commandbuffer.rs +192 -0
- package/native/tvos/metal-patched/src/commandqueue.rs +44 -0
- package/native/tvos/metal-patched/src/computepass.rs +107 -0
- package/native/tvos/metal-patched/src/constants.rs +152 -0
- package/native/tvos/metal-patched/src/counters.rs +119 -0
- package/native/tvos/metal-patched/src/depthstencil.rs +190 -0
- package/native/tvos/metal-patched/src/device.rs +2134 -0
- package/native/tvos/metal-patched/src/drawable.rs +39 -0
- package/native/tvos/metal-patched/src/encoder.rs +2041 -0
- package/native/tvos/metal-patched/src/heap.rs +281 -0
- package/native/tvos/metal-patched/src/indirect_encoder.rs +344 -0
- package/native/tvos/metal-patched/src/lib.rs +657 -0
- package/native/tvos/metal-patched/src/library.rs +902 -0
- package/native/tvos/metal-patched/src/mps.rs +575 -0
- package/native/tvos/metal-patched/src/pipeline/compute.rs +475 -0
- package/native/tvos/metal-patched/src/pipeline/mod.rs +71 -0
- package/native/tvos/metal-patched/src/pipeline/render.rs +762 -0
- package/native/tvos/metal-patched/src/renderpass.rs +443 -0
- package/native/tvos/metal-patched/src/resource.rs +182 -0
- package/native/tvos/metal-patched/src/sampler.rs +165 -0
- package/native/tvos/metal-patched/src/sync.rs +178 -0
- package/native/tvos/metal-patched/src/texture.rs +352 -0
- package/native/tvos/metal-patched/src/types.rs +90 -0
- package/native/tvos/metal-patched/src/vertexdescriptor.rs +250 -0
- package/native/tvos/src/audio_backend.rs +197 -0
- package/native/tvos/src/lib.rs +1891 -0
- package/native/visionos/Cargo.lock +1693 -0
- package/native/visionos/Cargo.toml +40 -0
- package/native/visionos/src/audio_backend.rs +197 -0
- package/native/visionos/src/lib.rs +1887 -0
- package/native/watchos/Cargo.lock +16 -0
- package/native/watchos/Cargo.toml +19 -0
- package/native/watchos/shaders/bloom_postfx.metal +99 -0
- package/native/watchos/src/BloomWatchApp.swift +1267 -0
- package/native/watchos/src/BloomWatchAudio.swift +179 -0
- package/native/watchos/src/audio.rs +55 -0
- package/native/watchos/src/draw_list.rs +229 -0
- package/native/watchos/src/ffi_stubs.rs +915 -0
- package/native/watchos/src/ffi_stubs_manual.rs +35 -0
- package/native/watchos/src/lib.rs +1124 -0
- package/native/watchos/src/models.rs +746 -0
- package/native/watchos/src/postfx.rs +95 -0
- package/native/watchos/src/scene.rs +534 -0
- package/native/watchos/src/textures.rs +184 -0
- package/native/web/Cargo.lock +1657 -0
- package/native/web/Cargo.toml +43 -0
- package/native/web/bloom_glue.js +695 -0
- package/native/web/build.sh +131 -0
- package/native/web/index.html +35 -0
- package/native/web/jolt_bridge.js +1519 -0
- package/native/web/src/input_ffi.rs +286 -0
- package/native/web/src/lib.rs +1796 -0
- package/native/web/src/material_ffi.rs +710 -0
- package/native/web/src/parity_ffi.rs +343 -0
- package/native/web/src/physics_ffi.rs +643 -0
- package/native/web/src/ragdoll_ffi.rs +250 -0
- package/native/web/src/render_settings.rs +98 -0
- package/native/windows/Cargo.lock +1815 -0
- package/native/windows/Cargo.toml +68 -0
- package/native/windows/src/lib.rs +1486 -0
- package/package.json +4279 -0
- package/src/audio/index.ts +315 -0
- package/src/core/colors.ts +63 -0
- package/src/core/index.ts +1206 -0
- package/src/core/keys.ts +63 -0
- package/src/core/types.ts +104 -0
- package/src/index.ts +171 -0
- package/src/math/index.ts +516 -0
- package/src/mobile/index.ts +294 -0
- package/src/models/index.ts +1258 -0
- package/src/physics/index.ts +1134 -0
- package/src/scene/index.ts +698 -0
- package/src/shapes/index.ts +120 -0
- package/src/text/index.ts +48 -0
- package/src/textures/index.ts +187 -0
- package/src/vfx/index.ts +191 -0
- package/src/world/index.ts +24 -0
- package/src/world/loader.ts +423 -0
- package/src/world/prefab.ts +217 -0
- package/src/world/render.ts +172 -0
- package/src/world/saver.ts +108 -0
- package/src/world/serialize.ts +301 -0
- package/src/world/terrain.ts +355 -0
- package/src/world/types.ts +160 -0
- package/src/world/validate.ts +319 -0
- package/src/world/version.ts +114 -0
|
@@ -0,0 +1,492 @@
|
|
|
1
|
+
//! Frame profiler: CPU phase timings + optional GPU timestamp queries.
|
|
2
|
+
//!
|
|
3
|
+
//! Disabled by default (zero overhead when `enabled == false`). Enable at
|
|
4
|
+
//! runtime via `Profiler::set_enabled(true)` — typically wired to a FFI
|
|
5
|
+
//! setter so games can toggle it from TS.
|
|
6
|
+
//!
|
|
7
|
+
//! CPU side: `begin("label")` / `end("label")` stacks around the phase.
|
|
8
|
+
//! GPU side: create pipeline using `timestamp_writes_for(&mut encoder, "label")`
|
|
9
|
+
//! and plug the returned `RenderPassTimestampWrites` into the pass descriptor.
|
|
10
|
+
//! GPU timestamps require the `TIMESTAMP_QUERY` device feature (opt-in at
|
|
11
|
+
//! device creation). When the feature is unavailable the GPU path no-ops
|
|
12
|
+
//! and only CPU numbers are reported.
|
|
13
|
+
//!
|
|
14
|
+
//! At `frame_end()` the profiler copies per-frame samples into a rolling
|
|
15
|
+
//! window (default 120 frames) and computes averages. `summary()` returns
|
|
16
|
+
//! a human-readable table.
|
|
17
|
+
//!
|
|
18
|
+
//! This module intentionally keeps the data model flat (Vec of named
|
|
19
|
+
//! samples, not a tree) — post-FX passes are already flat and a tree
|
|
20
|
+
//! would be overkill for a first pass.
|
|
21
|
+
|
|
22
|
+
#[cfg(feature = "web")]
|
|
23
|
+
use web_time::Instant;
|
|
24
|
+
#[cfg(not(feature = "web"))]
|
|
25
|
+
use std::time::Instant;
|
|
26
|
+
|
|
27
|
+
use std::collections::HashMap;
|
|
28
|
+
|
|
29
|
+
const ROLLING_FRAMES: usize = 120;
|
|
30
|
+
const MAX_GPU_PAIRS: u32 = 32;
|
|
31
|
+
|
|
32
|
+
#[derive(Clone, Copy)]
|
|
33
|
+
struct CpuSample {
|
|
34
|
+
start: Instant,
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
struct FrameSample {
|
|
38
|
+
label: &'static str,
|
|
39
|
+
cpu_us: f64,
|
|
40
|
+
gpu_us: Option<f64>,
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
pub struct Profiler {
|
|
44
|
+
pub enabled: bool,
|
|
45
|
+
pub gpu_enabled: bool,
|
|
46
|
+
|
|
47
|
+
open_cpu: HashMap<&'static str, CpuSample>,
|
|
48
|
+
frame: Vec<FrameSample>,
|
|
49
|
+
rolling: HashMap<&'static str, RollingStats>,
|
|
50
|
+
frame_count: u64,
|
|
51
|
+
|
|
52
|
+
// GPU timestamp query state
|
|
53
|
+
query_set: Option<wgpu::QuerySet>,
|
|
54
|
+
resolve_buffer: Option<wgpu::Buffer>,
|
|
55
|
+
readback_buffer: Option<wgpu::Buffer>,
|
|
56
|
+
timestamp_period_ns: f32,
|
|
57
|
+
next_query: u32,
|
|
58
|
+
// label -> (begin_index, end_index)
|
|
59
|
+
pending_gpu: Vec<(&'static str, u32, u32)>,
|
|
60
|
+
/// One-shot warning guard for GPU timestamp-pair exhaustion.
|
|
61
|
+
budget_warned: bool,
|
|
62
|
+
|
|
63
|
+
/// Phase 8 — last `ROLLING_FRAMES` frame totals (sum of all
|
|
64
|
+
/// samples in `frame` at frame_end), in microseconds. Ring
|
|
65
|
+
/// buffer indexed by `histogram_idx`; consumers pass through
|
|
66
|
+
/// `bloom_profiler_frame_history` and render a bar chart.
|
|
67
|
+
frame_total_cpu_us: [f64; ROLLING_FRAMES],
|
|
68
|
+
frame_total_gpu_us: [f64; ROLLING_FRAMES],
|
|
69
|
+
histogram_idx: usize,
|
|
70
|
+
histogram_filled: usize,
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
struct RollingStats {
|
|
74
|
+
cpu: [f64; ROLLING_FRAMES],
|
|
75
|
+
gpu: [f64; ROLLING_FRAMES],
|
|
76
|
+
has_gpu: bool,
|
|
77
|
+
idx: usize,
|
|
78
|
+
filled: usize,
|
|
79
|
+
/// Frame index of the most recent sample. A pass that stops running
|
|
80
|
+
/// (feature toggled off) must drop out of the readouts instead of
|
|
81
|
+
/// showing its frozen average forever.
|
|
82
|
+
last_frame: u64,
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
impl RollingStats {
|
|
86
|
+
fn new() -> Self {
|
|
87
|
+
Self { cpu: [0.0; ROLLING_FRAMES], gpu: [0.0; ROLLING_FRAMES], has_gpu: false, idx: 0, filled: 0, last_frame: 0 }
|
|
88
|
+
}
|
|
89
|
+
fn push(&mut self, cpu: f64, gpu: Option<f64>) {
|
|
90
|
+
self.cpu[self.idx] = cpu;
|
|
91
|
+
if let Some(g) = gpu { self.gpu[self.idx] = g; self.has_gpu = true; }
|
|
92
|
+
self.idx = (self.idx + 1) % ROLLING_FRAMES;
|
|
93
|
+
self.filled = (self.filled + 1).min(ROLLING_FRAMES);
|
|
94
|
+
}
|
|
95
|
+
fn avg_cpu(&self) -> f64 {
|
|
96
|
+
if self.filled == 0 { return 0.0; }
|
|
97
|
+
let sum: f64 = self.cpu.iter().take(self.filled).sum();
|
|
98
|
+
sum / self.filled as f64
|
|
99
|
+
}
|
|
100
|
+
fn avg_gpu(&self) -> Option<f64> {
|
|
101
|
+
if !self.has_gpu || self.filled == 0 { return None; }
|
|
102
|
+
let sum: f64 = self.gpu.iter().take(self.filled).sum();
|
|
103
|
+
Some(sum / self.filled as f64)
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
impl Profiler {
|
|
108
|
+
pub fn new() -> Self {
|
|
109
|
+
Self {
|
|
110
|
+
enabled: false,
|
|
111
|
+
gpu_enabled: false,
|
|
112
|
+
open_cpu: HashMap::new(),
|
|
113
|
+
frame: Vec::new(),
|
|
114
|
+
rolling: HashMap::new(),
|
|
115
|
+
frame_count: 0,
|
|
116
|
+
query_set: None,
|
|
117
|
+
resolve_buffer: None,
|
|
118
|
+
readback_buffer: None,
|
|
119
|
+
timestamp_period_ns: 1.0,
|
|
120
|
+
next_query: 0,
|
|
121
|
+
pending_gpu: Vec::new(),
|
|
122
|
+
budget_warned: false,
|
|
123
|
+
frame_total_cpu_us: [0.0; ROLLING_FRAMES],
|
|
124
|
+
frame_total_gpu_us: [0.0; ROLLING_FRAMES],
|
|
125
|
+
histogram_idx: 0,
|
|
126
|
+
histogram_filled: 0,
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/// Call after device creation. If the device supports `TIMESTAMP_QUERY`,
|
|
131
|
+
/// allocates the query set + readback buffer so GPU timings become available.
|
|
132
|
+
pub fn init_gpu(&mut self, device: &wgpu::Device, queue: &wgpu::Queue) {
|
|
133
|
+
if !device.features().contains(wgpu::Features::TIMESTAMP_QUERY) {
|
|
134
|
+
return;
|
|
135
|
+
}
|
|
136
|
+
let query_set = device.create_query_set(&wgpu::QuerySetDescriptor {
|
|
137
|
+
label: Some("bloom_profiler_queryset"),
|
|
138
|
+
ty: wgpu::QueryType::Timestamp,
|
|
139
|
+
count: MAX_GPU_PAIRS * 2,
|
|
140
|
+
});
|
|
141
|
+
let resolve_size = (MAX_GPU_PAIRS as u64) * 2 * 8;
|
|
142
|
+
let resolve_buffer = device.create_buffer(&wgpu::BufferDescriptor {
|
|
143
|
+
label: Some("bloom_profiler_resolve"),
|
|
144
|
+
size: resolve_size,
|
|
145
|
+
usage: wgpu::BufferUsages::QUERY_RESOLVE | wgpu::BufferUsages::COPY_SRC,
|
|
146
|
+
mapped_at_creation: false,
|
|
147
|
+
});
|
|
148
|
+
let readback_buffer = device.create_buffer(&wgpu::BufferDescriptor {
|
|
149
|
+
label: Some("bloom_profiler_readback"),
|
|
150
|
+
size: resolve_size,
|
|
151
|
+
usage: wgpu::BufferUsages::MAP_READ | wgpu::BufferUsages::COPY_DST,
|
|
152
|
+
mapped_at_creation: false,
|
|
153
|
+
});
|
|
154
|
+
self.query_set = Some(query_set);
|
|
155
|
+
self.resolve_buffer = Some(resolve_buffer);
|
|
156
|
+
self.readback_buffer = Some(readback_buffer);
|
|
157
|
+
self.timestamp_period_ns = queue.get_timestamp_period();
|
|
158
|
+
self.gpu_enabled = true;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
pub fn set_enabled(&mut self, on: bool) {
|
|
162
|
+
if on && !self.enabled {
|
|
163
|
+
// Fresh measuring session: without this, stats captured before
|
|
164
|
+
// a disable (potentially under a different feature set) show
|
|
165
|
+
// until the rolling window refills and skew the first seconds
|
|
166
|
+
// of every new session.
|
|
167
|
+
self.rolling.clear();
|
|
168
|
+
self.frame_total_cpu_us = [0.0; ROLLING_FRAMES];
|
|
169
|
+
self.frame_total_gpu_us = [0.0; ROLLING_FRAMES];
|
|
170
|
+
self.histogram_idx = 0;
|
|
171
|
+
self.histogram_filled = 0;
|
|
172
|
+
}
|
|
173
|
+
self.enabled = on;
|
|
174
|
+
}
|
|
175
|
+
pub fn is_enabled(&self) -> bool { self.enabled }
|
|
176
|
+
pub fn has_gpu(&self) -> bool { self.gpu_enabled }
|
|
177
|
+
|
|
178
|
+
pub fn begin(&mut self, label: &'static str) {
|
|
179
|
+
if !self.enabled { return; }
|
|
180
|
+
self.open_cpu.insert(label, CpuSample { start: Instant::now() });
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
pub fn end(&mut self, label: &'static str) {
|
|
184
|
+
if !self.enabled { return; }
|
|
185
|
+
if let Some(s) = self.open_cpu.remove(label) {
|
|
186
|
+
let us = s.start.elapsed().as_secs_f64() * 1_000_000.0;
|
|
187
|
+
self.frame.push(FrameSample { label, cpu_us: us, gpu_us: None });
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/// Reserve a pair of GPU timestamp slot indices for the given label.
|
|
192
|
+
/// Callers combine the returned indices with `query_set()` to build a
|
|
193
|
+
/// `RenderPassTimestampWrites` for the pass descriptor. Returns None
|
|
194
|
+
/// when profiling or GPU queries are disabled, or when no slots remain.
|
|
195
|
+
pub fn reserve_gpu_pair(&mut self, label: &'static str) -> Option<(u32, u32)> {
|
|
196
|
+
if !self.enabled || !self.gpu_enabled { return None; }
|
|
197
|
+
self.query_set.as_ref()?;
|
|
198
|
+
if self.next_query + 2 > MAX_GPU_PAIRS * 2 {
|
|
199
|
+
if !self.budget_warned {
|
|
200
|
+
self.budget_warned = true;
|
|
201
|
+
eprintln!(
|
|
202
|
+
"bloom profiler: GPU timestamp budget ({} pairs) exhausted — later passes report CPU time only",
|
|
203
|
+
MAX_GPU_PAIRS
|
|
204
|
+
);
|
|
205
|
+
}
|
|
206
|
+
return None;
|
|
207
|
+
}
|
|
208
|
+
let begin = self.next_query;
|
|
209
|
+
let end = self.next_query + 1;
|
|
210
|
+
self.next_query += 2;
|
|
211
|
+
self.pending_gpu.push((label, begin, end));
|
|
212
|
+
Some((begin, end))
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
pub fn query_set(&self) -> Option<&wgpu::QuerySet> {
|
|
216
|
+
self.query_set.as_ref()
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/// Convenience: build a `RenderPassTimestampWrites` for a pass in one call.
|
|
220
|
+
/// Returns None when GPU queries are disabled — the caller should plug
|
|
221
|
+
/// the returned `Option` straight into `RenderPassDescriptor.timestamp_writes`.
|
|
222
|
+
pub fn pass_timestamp_writes(&mut self, label: &'static str) -> Option<wgpu::RenderPassTimestampWrites<'_>> {
|
|
223
|
+
let (b, e) = self.reserve_gpu_pair(label)?;
|
|
224
|
+
let qs = self.query_set.as_ref()?;
|
|
225
|
+
Some(wgpu::RenderPassTimestampWrites {
|
|
226
|
+
query_set: qs,
|
|
227
|
+
beginning_of_pass_write_index: Some(b),
|
|
228
|
+
end_of_pass_write_index: Some(e),
|
|
229
|
+
})
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/// Same as `pass_timestamp_writes` but for a compute pass descriptor.
|
|
233
|
+
pub fn compute_pass_timestamp_writes(&mut self, label: &'static str) -> Option<wgpu::ComputePassTimestampWrites<'_>> {
|
|
234
|
+
let (b, e) = self.reserve_gpu_pair(label)?;
|
|
235
|
+
let qs = self.query_set.as_ref()?;
|
|
236
|
+
Some(wgpu::ComputePassTimestampWrites {
|
|
237
|
+
query_set: qs,
|
|
238
|
+
beginning_of_pass_write_index: Some(b),
|
|
239
|
+
end_of_pass_write_index: Some(e),
|
|
240
|
+
})
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
/// Resolve any pending GPU queries into the readback buffer. Call once
|
|
244
|
+
/// per frame, after all passes are encoded and before submit.
|
|
245
|
+
pub fn resolve(&mut self, encoder: &mut wgpu::CommandEncoder) {
|
|
246
|
+
if !self.enabled || !self.gpu_enabled || self.next_query == 0 { return; }
|
|
247
|
+
let (Some(qs), Some(resolve)) = (&self.query_set, &self.resolve_buffer) else { return; };
|
|
248
|
+
encoder.resolve_query_set(qs, 0..self.next_query, resolve, 0);
|
|
249
|
+
if let Some(readback) = &self.readback_buffer {
|
|
250
|
+
let byte_count = (self.next_query as u64) * 8;
|
|
251
|
+
encoder.copy_buffer_to_buffer(resolve, 0, readback, 0, byte_count);
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
/// End-of-frame bookkeeping. Resolves this frame's GPU timestamps via
|
|
256
|
+
/// a BLOCKING map (map_async + poll(Wait) — serialises CPU⇄GPU, so
|
|
257
|
+
/// wall-clock fps is pessimistic while the profiler is enabled; see
|
|
258
|
+
/// docs/crash-triage-windows.md), folds samples into rolling stats,
|
|
259
|
+
/// and clears per-frame state for the next frame.
|
|
260
|
+
pub fn frame_end(&mut self, device: &wgpu::Device) {
|
|
261
|
+
if !self.enabled {
|
|
262
|
+
self.frame.clear();
|
|
263
|
+
self.open_cpu.clear();
|
|
264
|
+
self.next_query = 0;
|
|
265
|
+
self.pending_gpu.clear();
|
|
266
|
+
return;
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
if self.gpu_enabled && self.next_query > 0 {
|
|
270
|
+
if let Some(readback) = &self.readback_buffer {
|
|
271
|
+
let byte_count = (self.next_query as u64) * 8;
|
|
272
|
+
let slice = readback.slice(0..byte_count);
|
|
273
|
+
slice.map_async(wgpu::MapMode::Read, |_| {});
|
|
274
|
+
let _ = device.poll(wgpu::PollType::Wait { submission_index: None, timeout: None });
|
|
275
|
+
let data = slice.get_mapped_range().to_vec();
|
|
276
|
+
readback.unmap();
|
|
277
|
+
let period = self.timestamp_period_ns as f64;
|
|
278
|
+
let mut by_label: HashMap<&'static str, f64> = HashMap::new();
|
|
279
|
+
for (label, b, e) in &self.pending_gpu {
|
|
280
|
+
let bo = (*b as usize) * 8;
|
|
281
|
+
let eo = (*e as usize) * 8;
|
|
282
|
+
if eo + 8 > data.len() { continue; }
|
|
283
|
+
let bt = u64::from_le_bytes(data[bo..bo+8].try_into().unwrap());
|
|
284
|
+
let et = u64::from_le_bytes(data[eo..eo+8].try_into().unwrap());
|
|
285
|
+
if et <= bt { continue; }
|
|
286
|
+
let us = (et - bt) as f64 * period / 1000.0;
|
|
287
|
+
*by_label.entry(*label).or_insert(0.0) += us;
|
|
288
|
+
}
|
|
289
|
+
for s in self.frame.iter_mut() {
|
|
290
|
+
if let Some(us) = by_label.remove(s.label) { s.gpu_us = Some(us); }
|
|
291
|
+
}
|
|
292
|
+
// GPU samples without a CPU counterpart — record them too.
|
|
293
|
+
for (label, us) in by_label {
|
|
294
|
+
self.frame.push(FrameSample { label, cpu_us: 0.0, gpu_us: Some(us) });
|
|
295
|
+
}
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
self.frame_end_cpu();
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/// CPU-only end-of-frame: histogram update + drain into rolling.
|
|
303
|
+
/// Split out so tests don't need a wgpu::Device. Production
|
|
304
|
+
/// callers go through `frame_end` which handles GPU readback
|
|
305
|
+
/// first and then delegates here.
|
|
306
|
+
fn frame_end_cpu(&mut self) {
|
|
307
|
+
// Phase 8 — sum the per-pass samples into a per-frame total
|
|
308
|
+
// for the histogram before draining. CPU sums every sample;
|
|
309
|
+
// GPU sums only those with timestamps (rest have None).
|
|
310
|
+
let mut frame_cpu = 0.0;
|
|
311
|
+
let mut frame_gpu = 0.0;
|
|
312
|
+
for s in &self.frame {
|
|
313
|
+
frame_cpu += s.cpu_us;
|
|
314
|
+
if let Some(g) = s.gpu_us { frame_gpu += g; }
|
|
315
|
+
}
|
|
316
|
+
self.frame_total_cpu_us[self.histogram_idx] = frame_cpu;
|
|
317
|
+
self.frame_total_gpu_us[self.histogram_idx] = frame_gpu;
|
|
318
|
+
self.histogram_idx = (self.histogram_idx + 1) % ROLLING_FRAMES;
|
|
319
|
+
self.histogram_filled = (self.histogram_filled + 1).min(ROLLING_FRAMES);
|
|
320
|
+
|
|
321
|
+
let fc = self.frame_count;
|
|
322
|
+
for s in self.frame.drain(..) {
|
|
323
|
+
let entry = self.rolling.entry(s.label).or_insert_with(RollingStats::new);
|
|
324
|
+
entry.push(s.cpu_us, s.gpu_us);
|
|
325
|
+
entry.last_frame = fc;
|
|
326
|
+
}
|
|
327
|
+
// Drop entries that have not reported for several windows so a
|
|
328
|
+
// disabled feature's passes leave the map instead of lingering.
|
|
329
|
+
self.rolling.retain(|_, s| fc.saturating_sub(s.last_frame) <= (4 * ROLLING_FRAMES) as u64);
|
|
330
|
+
self.open_cpu.clear();
|
|
331
|
+
self.next_query = 0;
|
|
332
|
+
self.pending_gpu.clear();
|
|
333
|
+
self.frame_count = self.frame_count.wrapping_add(1);
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
/// Phase 8 — frame-history snapshot for the overlay's bar chart.
|
|
337
|
+
/// Returns `(cpu_us, gpu_us)` pairs in chronological order, oldest
|
|
338
|
+
/// first, exactly `ROLLING_FRAMES.min(filled)` entries long.
|
|
339
|
+
pub fn frame_history(&self) -> Vec<(f64, f64)> {
|
|
340
|
+
let n = self.histogram_filled;
|
|
341
|
+
if n == 0 { return Vec::new(); }
|
|
342
|
+
let start = if n < ROLLING_FRAMES { 0 } else { self.histogram_idx };
|
|
343
|
+
let mut out = Vec::with_capacity(n);
|
|
344
|
+
for i in 0..n {
|
|
345
|
+
let idx = (start + i) % ROLLING_FRAMES;
|
|
346
|
+
out.push((self.frame_total_cpu_us[idx], self.frame_total_gpu_us[idx]));
|
|
347
|
+
}
|
|
348
|
+
out
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
pub fn summary(&self) -> String {
|
|
352
|
+
if !self.enabled {
|
|
353
|
+
return String::from("profiler: disabled\n");
|
|
354
|
+
}
|
|
355
|
+
let fc = self.frame_count;
|
|
356
|
+
let mut entries: Vec<(&&str, &RollingStats)> = self.rolling.iter()
|
|
357
|
+
.filter(|(_, s)| fc.saturating_sub(s.last_frame) <= ROLLING_FRAMES as u64)
|
|
358
|
+
.map(|(k, v)| (k, v))
|
|
359
|
+
.collect();
|
|
360
|
+
entries.sort_by(|a, b| b.1.avg_cpu().partial_cmp(&a.1.avg_cpu()).unwrap_or(std::cmp::Ordering::Equal));
|
|
361
|
+
|
|
362
|
+
let mut out = String::new();
|
|
363
|
+
out.push_str(&format!(
|
|
364
|
+
"profiler (avg over last {} frames, gpu={}):\n",
|
|
365
|
+
ROLLING_FRAMES.min(entries.first().map(|(_,s)| s.filled).unwrap_or(0)),
|
|
366
|
+
if self.gpu_enabled { "yes" } else { "no" },
|
|
367
|
+
));
|
|
368
|
+
out.push_str(" phase cpu us gpu us\n");
|
|
369
|
+
for (label, stats) in entries {
|
|
370
|
+
let gpu = stats.avg_gpu().map(|v| format!("{:>9.1}", v)).unwrap_or_else(|| " -".to_string());
|
|
371
|
+
out.push_str(&format!(" {:<28} {:>9.1} {}\n", label, stats.avg_cpu(), gpu));
|
|
372
|
+
}
|
|
373
|
+
out
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
/// Average total CPU frame time across the rolling window (sum of
|
|
377
|
+
/// all phases). Useful for a single headline number.
|
|
378
|
+
pub fn avg_frame_cpu_us(&self) -> f64 {
|
|
379
|
+
let fc = self.frame_count;
|
|
380
|
+
self.rolling.values()
|
|
381
|
+
.filter(|s| fc.saturating_sub(s.last_frame) <= ROLLING_FRAMES as u64)
|
|
382
|
+
.map(|s| s.avg_cpu()).sum()
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
/// Average total GPU frame time where available.
|
|
386
|
+
pub fn avg_frame_gpu_us(&self) -> f64 {
|
|
387
|
+
let fc = self.frame_count;
|
|
388
|
+
self.rolling.values()
|
|
389
|
+
.filter(|s| fc.saturating_sub(s.last_frame) <= ROLLING_FRAMES as u64)
|
|
390
|
+
.filter_map(|s| s.avg_gpu()).sum()
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
/// Snapshot the rolling averages in a stable, CPU-time-descending
|
|
394
|
+
/// order. Games call this once per overlay-draw frame and pull
|
|
395
|
+
/// label/cpu/gpu out via the accessors below — HashMap iteration
|
|
396
|
+
/// order would jitter the overlay otherwise.
|
|
397
|
+
pub fn snapshot(&mut self) -> Vec<(&'static str, f64, Option<f64>)> {
|
|
398
|
+
let fc = self.frame_count;
|
|
399
|
+
let mut v: Vec<(&'static str, f64, Option<f64>)> = self.rolling.iter()
|
|
400
|
+
.filter(|(_, s)| fc.saturating_sub(s.last_frame) <= ROLLING_FRAMES as u64)
|
|
401
|
+
.map(|(k, s)| (*k, s.avg_cpu(), s.avg_gpu()))
|
|
402
|
+
.collect();
|
|
403
|
+
v.sort_by(|a, b| b.1.partial_cmp(&a.1).unwrap_or(std::cmp::Ordering::Equal));
|
|
404
|
+
v
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
#[cfg(test)]
|
|
409
|
+
mod tests {
|
|
410
|
+
use super::*;
|
|
411
|
+
|
|
412
|
+
/// Drive the profiler through `frame_end` purely on the CPU
|
|
413
|
+
/// side (gpu_enabled stays false). Each "frame" pushes a single
|
|
414
|
+
/// sample with the given cpu_us so the histogram totals are
|
|
415
|
+
/// known.
|
|
416
|
+
fn fake_frame(p: &mut Profiler, cpu_us: f64) {
|
|
417
|
+
p.frame.push(FrameSample { label: "fake", cpu_us, gpu_us: None });
|
|
418
|
+
// Skip the gpu readback path (no device available) and go
|
|
419
|
+
// straight to the CPU-only histogram + drain path.
|
|
420
|
+
p.frame_end_cpu();
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
#[test]
|
|
424
|
+
fn frame_history_empty_before_any_frames() {
|
|
425
|
+
let mut p = Profiler::new();
|
|
426
|
+
p.set_enabled(true);
|
|
427
|
+
assert!(p.frame_history().is_empty());
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
#[test]
|
|
431
|
+
fn frame_history_records_in_order_under_capacity() {
|
|
432
|
+
let mut p = Profiler::new();
|
|
433
|
+
p.set_enabled(true);
|
|
434
|
+
for i in 1..=5 {
|
|
435
|
+
fake_frame(&mut p, i as f64 * 100.0);
|
|
436
|
+
}
|
|
437
|
+
let h = p.frame_history();
|
|
438
|
+
assert_eq!(h.len(), 5);
|
|
439
|
+
assert_eq!(h[0].0, 100.0);
|
|
440
|
+
assert_eq!(h[4].0, 500.0);
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
#[test]
|
|
444
|
+
fn stale_labels_drop_out_of_snapshot_and_evict() {
|
|
445
|
+
let mut p = Profiler::new();
|
|
446
|
+
p.set_enabled(true);
|
|
447
|
+
// One label reports once, another keeps reporting.
|
|
448
|
+
p.frame.push(FrameSample { label: "once", cpu_us: 5.0, gpu_us: None });
|
|
449
|
+
p.frame_end_cpu();
|
|
450
|
+
for _ in 0..(ROLLING_FRAMES + 1) { fake_frame(&mut p, 1.0); }
|
|
451
|
+
let snap = p.snapshot();
|
|
452
|
+
assert!(snap.iter().any(|(l, _, _)| *l == "fake"));
|
|
453
|
+
assert!(
|
|
454
|
+
!snap.iter().any(|(l, _, _)| *l == "once"),
|
|
455
|
+
"a pass that stopped reporting must leave the snapshot"
|
|
456
|
+
);
|
|
457
|
+
for _ in 0..(4 * ROLLING_FRAMES) { fake_frame(&mut p, 1.0); }
|
|
458
|
+
assert!(!p.rolling.contains_key("once"), "stale label must eventually evict");
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
#[test]
|
|
462
|
+
fn reenable_starts_fresh_session() {
|
|
463
|
+
let mut p = Profiler::new();
|
|
464
|
+
p.set_enabled(true);
|
|
465
|
+
fake_frame(&mut p, 100.0);
|
|
466
|
+
assert!(!p.frame_history().is_empty());
|
|
467
|
+
p.set_enabled(false);
|
|
468
|
+
p.set_enabled(true);
|
|
469
|
+
assert!(p.frame_history().is_empty(), "histogram must clear on re-enable");
|
|
470
|
+
assert!(p.snapshot().is_empty(), "rolling stats must clear on re-enable");
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
#[test]
|
|
474
|
+
fn frame_history_wraps_oldest_first_at_capacity() {
|
|
475
|
+
let mut p = Profiler::new();
|
|
476
|
+
p.set_enabled(true);
|
|
477
|
+
// Push more than ROLLING_FRAMES so the ring wraps.
|
|
478
|
+
for i in 1..=(ROLLING_FRAMES + 30) {
|
|
479
|
+
fake_frame(&mut p, i as f64);
|
|
480
|
+
}
|
|
481
|
+
let h = p.frame_history();
|
|
482
|
+
assert_eq!(h.len(), ROLLING_FRAMES);
|
|
483
|
+
// The newest entry is whatever the last push was.
|
|
484
|
+
assert_eq!(h.last().unwrap().0, (ROLLING_FRAMES + 30) as f64);
|
|
485
|
+
// The oldest must be `(ROLLING_FRAMES + 30) - ROLLING_FRAMES + 1 = 31`.
|
|
486
|
+
assert_eq!(h[0].0, 31.0);
|
|
487
|
+
// Strictly monotonic (oldest → newest).
|
|
488
|
+
for w in h.windows(2) {
|
|
489
|
+
assert!(w[0].0 < w[1].0, "history must be in chronological order");
|
|
490
|
+
}
|
|
491
|
+
}
|
|
492
|
+
}
|