From 4212839f0dd8acd62821b264902e9976dbc2dbfd Mon Sep 17 00:00:00 2001 From: supersurviveur Date: Sun, 27 Sep 2026 14:05:09 +0200 Subject: [PATCH] =?UTF-8?q?fix:=E2=80=AFavoid=20hitting=20GPU=E2=80=AFwatc?= =?UTF-8?q?hdog?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- src/fractal/renderer.rs | 106 ++++++++++++++++++++++++++++++++++++++-- 1 file changed, 101 insertions(+), 5 deletions(-) diff --git a/src/fractal/renderer.rs b/src/fractal/renderer.rs index 59fd37b..21759e2 100644 --- a/src/fractal/renderer.rs +++ b/src/fractal/renderer.rs @@ -1309,8 +1309,16 @@ impl ExportRender { }); // ~128px bands, kept to a sane range so progress is smooth without too - // many submissions. - let tiles = (height / 128).clamp(8, 64).min(height.max(1)); + // many submissions; more at high iteration counts so no single + // submission runs long enough to trip a GPU reset (see + // `WORK_PER_SUBMIT`). Export supersamples every pixel (×3 in shadow + // mode, which also iterates two neighbours). + let aa = uniforms.aa_level.max(1); + let samples = aa * aa * if uniforms.rendering_mode != 0 { 3 } else { 1 }; + let tiles = (height / 128) + .clamp(8, 64) + .max(band_count(width, height, uniforms.max_iter, samples)) + .min(height.max(1)); let swap_rb = matches!( target_format, @@ -1758,6 +1766,25 @@ fn data_pass( pipeline: &wgpu::RenderPipeline, bind_groups: &[&wgpu::BindGroup], ) { + band_pass(encoder, label, target, pipeline, bind_groups, None, true); +} + +/// [`data_pass`] restricted to `rows` (`[y0, y1)` of a `width`-wide target), +/// clearing the whole attachment first only if `clear`. +fn band_pass( + encoder: &mut wgpu::CommandEncoder, + label: &str, + target: &wgpu::TextureView, + pipeline: &wgpu::RenderPipeline, + bind_groups: &[&wgpu::BindGroup], + rows: Option<(u32, u32, u32)>, + clear: bool, +) { + let load = if clear { + wgpu::LoadOp::Clear(wgpu::Color::BLACK) + } else { + wgpu::LoadOp::Load + }; let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor { label: Some(label), color_attachments: &[Some(wgpu::RenderPassColorAttachment { @@ -1765,7 +1792,7 @@ fn data_pass( depth_slice: None, resolve_target: None, ops: wgpu::Operations { - load: wgpu::LoadOp::Clear(wgpu::Color::BLACK), + load, store: wgpu::StoreOp::Store, }, })], @@ -1774,6 +1801,11 @@ fn data_pass( occlusion_query_set: None, multiview_mask: None, }); + if let Some((width, y0, y1)) = rows { + // Full-viewport triangle (so the pixel→plane mapping is unchanged), + // scissored to the band. + pass.set_scissor_rect(0, y0, width, y1 - y0); + } pass.set_pipeline(pipeline); for (i, bg) in bind_groups.iter().enumerate() { pass.set_bind_group(i as u32, *bg, &[]); @@ -1781,6 +1813,60 @@ fn data_pass( pass.draw(0..3, 0..1); } +/// Worst-case pixel·iteration work allowed in one GPU submission. Drivers +/// reset the GPU when a single draw runs too long (i915 on integrated Intel +/// gives up after ~640 ms when it can't preempt, and a fullscreen draw can't +/// be preempted mid-way), which loses the device. At deep zooms the iteration +/// count reaches millions, so iteration passes are split into row bands, each +/// its own submission, so the driver can schedule other work in between. +/// 2^30 is a few tens of ms at worst on an integrated GPU. +const WORK_PER_SUBMIT: f64 = (1u64 << 30) as f64; + +/// Number of row bands a `width`×`height` pass of up to `samples` × +/// `max_iter` iterations per pixel needs to stay within [`WORK_PER_SUBMIT`] +/// each (at most one band per row). +fn band_count(width: u32, height: u32, max_iter: u32, samples: u32) -> u32 { + let work = width as f64 * height as f64 * max_iter.max(1) as f64 * samples.max(1) as f64; + ((work / WORK_PER_SUBMIT).ceil() as u32).clamp(1, height.max(1)) +} + +/// [`data_pass`] split into `bands` row bands. One band is recorded into +/// `encoder` as usual; more are each submitted on their own right away (so +/// they run before `encoder`, which is submitted later and reads the result). +#[allow(clippy::too_many_arguments)] +fn banded_data_pass( + device: &wgpu::Device, + queue: &wgpu::Queue, + encoder: &mut wgpu::CommandEncoder, + label: &str, + target: &wgpu::TextureView, + pipeline: &wgpu::RenderPipeline, + bind_groups: &[&wgpu::BindGroup], + [width, height]: [u32; 2], + bands: u32, +) { + if bands <= 1 { + data_pass(encoder, label, target, pipeline, bind_groups); + return; + } + let band = height.div_ceil(bands); + for y0 in (0..height).step_by(band as usize) { + let y1 = (y0 + band).min(height); + let mut band_encoder = + device.create_command_encoder(&wgpu::CommandEncoderDescriptor { label: Some(label) }); + band_pass( + &mut band_encoder, + label, + target, + pipeline, + bind_groups, + Some((width, y0, y1)), + y0 == 0, + ); + queue.submit([band_encoder.finish()]); + } +} + /// A per-frame paint callback. Carries this frame's uniforms plus a reference to /// the current reference orbit (cheap `Arc` clone). The orbit is only re-uploaded /// when its `generation` changes; the expensive iteration pass re-runs only when @@ -1907,22 +1993,32 @@ impl egui_wgpu::CallbackTrait for FractalCallback { let pipelines = &renderer.pipelines[&PipelineKey::from_uniforms(&self.uniforms)]; if let Some(cache) = &renderer.cache { if iter_dirty { + let max_iter = self.uniforms.max_iter; // Iteration pass: 1-spp perturbation iterate → data texture. - data_pass( + banded_data_pass( + device, + queue, egui_encoder, "fractal iterate pass", &cache.data_view, &pipelines.iterate, &[&renderer.bind_group], + [width, height], + band_count(width, height, max_iter, 1), ); if let Some((data_aa_view, _)) = &cache.aa { // Adaptive AA: supersample only the non-smooth pixels. - data_pass( + let aa = self.uniforms.aa_level; + banded_data_pass( + device, + queue, egui_encoder, "fractal AA refine pass", data_aa_view, &pipelines.refine, &[&renderer.bind_group, &cache.refine_bind_group], + [width, height], + band_count(width, height, max_iter, aa * aa), ); } }