fix: avoid hitting GPU watchdog

This commit is contained in:
2026-09-27 14:05:09 +02:00
parent 48560de315
commit 4212839f0d
+101 -5
View File
@@ -1309,8 +1309,16 @@ impl ExportRender {
}); });
// ~128px bands, kept to a sane range so progress is smooth without too // ~128px bands, kept to a sane range so progress is smooth without too
// many submissions. // many submissions; more at high iteration counts so no single
let tiles = (height / 128).clamp(8, 64).min(height.max(1)); // submission runs long enough to trip a GPU reset (see
// `WORK_PER_SUBMIT`). Export supersamples every pixel (×3 in shadow
// mode, which also iterates two neighbours).
let aa = uniforms.aa_level.max(1);
let samples = aa * aa * if uniforms.rendering_mode != 0 { 3 } else { 1 };
let tiles = (height / 128)
.clamp(8, 64)
.max(band_count(width, height, uniforms.max_iter, samples))
.min(height.max(1));
let swap_rb = matches!( let swap_rb = matches!(
target_format, target_format,
@@ -1758,6 +1766,25 @@ fn data_pass(
pipeline: &wgpu::RenderPipeline, pipeline: &wgpu::RenderPipeline,
bind_groups: &[&wgpu::BindGroup], bind_groups: &[&wgpu::BindGroup],
) { ) {
band_pass(encoder, label, target, pipeline, bind_groups, None, true);
}
/// [`data_pass`] restricted to `rows` (`[y0, y1)` of a `width`-wide target),
/// clearing the whole attachment first only if `clear`.
fn band_pass(
encoder: &mut wgpu::CommandEncoder,
label: &str,
target: &wgpu::TextureView,
pipeline: &wgpu::RenderPipeline,
bind_groups: &[&wgpu::BindGroup],
rows: Option<(u32, u32, u32)>,
clear: bool,
) {
let load = if clear {
wgpu::LoadOp::Clear(wgpu::Color::BLACK)
} else {
wgpu::LoadOp::Load
};
let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor { let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
label: Some(label), label: Some(label),
color_attachments: &[Some(wgpu::RenderPassColorAttachment { color_attachments: &[Some(wgpu::RenderPassColorAttachment {
@@ -1765,7 +1792,7 @@ fn data_pass(
depth_slice: None, depth_slice: None,
resolve_target: None, resolve_target: None,
ops: wgpu::Operations { ops: wgpu::Operations {
load: wgpu::LoadOp::Clear(wgpu::Color::BLACK), load,
store: wgpu::StoreOp::Store, store: wgpu::StoreOp::Store,
}, },
})], })],
@@ -1774,6 +1801,11 @@ fn data_pass(
occlusion_query_set: None, occlusion_query_set: None,
multiview_mask: None, multiview_mask: None,
}); });
if let Some((width, y0, y1)) = rows {
// Full-viewport triangle (so the pixel→plane mapping is unchanged),
// scissored to the band.
pass.set_scissor_rect(0, y0, width, y1 - y0);
}
pass.set_pipeline(pipeline); pass.set_pipeline(pipeline);
for (i, bg) in bind_groups.iter().enumerate() { for (i, bg) in bind_groups.iter().enumerate() {
pass.set_bind_group(i as u32, *bg, &[]); pass.set_bind_group(i as u32, *bg, &[]);
@@ -1781,6 +1813,60 @@ fn data_pass(
pass.draw(0..3, 0..1); pass.draw(0..3, 0..1);
} }
/// Worst-case pixel·iteration work allowed in one GPU submission. Drivers
/// reset the GPU when a single draw runs too long (i915 on integrated Intel
/// gives up after ~640 ms when it can't preempt, and a fullscreen draw can't
/// be preempted mid-way), which loses the device. At deep zooms the iteration
/// count reaches millions, so iteration passes are split into row bands, each
/// its own submission, so the driver can schedule other work in between.
/// 2^30 is a few tens of ms at worst on an integrated GPU.
const WORK_PER_SUBMIT: f64 = (1u64 << 30) as f64;
/// Number of row bands a `width`×`height` pass of up to `samples` ×
/// `max_iter` iterations per pixel needs to stay within [`WORK_PER_SUBMIT`]
/// each (at most one band per row).
fn band_count(width: u32, height: u32, max_iter: u32, samples: u32) -> u32 {
let work = width as f64 * height as f64 * max_iter.max(1) as f64 * samples.max(1) as f64;
((work / WORK_PER_SUBMIT).ceil() as u32).clamp(1, height.max(1))
}
/// [`data_pass`] split into `bands` row bands. One band is recorded into
/// `encoder` as usual; more are each submitted on their own right away (so
/// they run before `encoder`, which is submitted later and reads the result).
#[allow(clippy::too_many_arguments)]
fn banded_data_pass(
device: &wgpu::Device,
queue: &wgpu::Queue,
encoder: &mut wgpu::CommandEncoder,
label: &str,
target: &wgpu::TextureView,
pipeline: &wgpu::RenderPipeline,
bind_groups: &[&wgpu::BindGroup],
[width, height]: [u32; 2],
bands: u32,
) {
if bands <= 1 {
data_pass(encoder, label, target, pipeline, bind_groups);
return;
}
let band = height.div_ceil(bands);
for y0 in (0..height).step_by(band as usize) {
let y1 = (y0 + band).min(height);
let mut band_encoder =
device.create_command_encoder(&wgpu::CommandEncoderDescriptor { label: Some(label) });
band_pass(
&mut band_encoder,
label,
target,
pipeline,
bind_groups,
Some((width, y0, y1)),
y0 == 0,
);
queue.submit([band_encoder.finish()]);
}
}
/// A per-frame paint callback. Carries this frame's uniforms plus a reference to /// A per-frame paint callback. Carries this frame's uniforms plus a reference to
/// the current reference orbit (cheap `Arc` clone). The orbit is only re-uploaded /// the current reference orbit (cheap `Arc` clone). The orbit is only re-uploaded
/// when its `generation` changes; the expensive iteration pass re-runs only when /// when its `generation` changes; the expensive iteration pass re-runs only when
@@ -1907,22 +1993,32 @@ impl egui_wgpu::CallbackTrait for FractalCallback {
let pipelines = &renderer.pipelines[&PipelineKey::from_uniforms(&self.uniforms)]; let pipelines = &renderer.pipelines[&PipelineKey::from_uniforms(&self.uniforms)];
if let Some(cache) = &renderer.cache { if let Some(cache) = &renderer.cache {
if iter_dirty { if iter_dirty {
let max_iter = self.uniforms.max_iter;
// Iteration pass: 1-spp perturbation iterate → data texture. // Iteration pass: 1-spp perturbation iterate → data texture.
data_pass( banded_data_pass(
device,
queue,
egui_encoder, egui_encoder,
"fractal iterate pass", "fractal iterate pass",
&cache.data_view, &cache.data_view,
&pipelines.iterate, &pipelines.iterate,
&[&renderer.bind_group], &[&renderer.bind_group],
[width, height],
band_count(width, height, max_iter, 1),
); );
if let Some((data_aa_view, _)) = &cache.aa { if let Some((data_aa_view, _)) = &cache.aa {
// Adaptive AA: supersample only the non-smooth pixels. // Adaptive AA: supersample only the non-smooth pixels.
data_pass( let aa = self.uniforms.aa_level;
banded_data_pass(
device,
queue,
egui_encoder, egui_encoder,
"fractal AA refine pass", "fractal AA refine pass",
data_aa_view, data_aa_view,
&pipelines.refine, &pipelines.refine,
&[&renderer.bind_group, &cache.refine_bind_group], &[&renderer.bind_group, &cache.refine_bind_group],
[width, height],
band_count(width, height, max_iter, aa * aa),
); );
} }
} }