perf: improve overall performance

This commit is contained in:
2026-09-24 18:23:19 +02:00
parent 7dd99cc1af
commit a3fd152dff
13 changed files with 1070 additions and 412 deletions
+35 -7
View File
@@ -85,7 +85,13 @@ pixel is a handful of `f32` complex multiplies.
`ALL` array used to enumerate every kind.
- `src/fractal/reference.rs` — `compute_reference`/`compute_set_reference`:
iterate the chosen formula at high precision on the CPU, emitting `Z_n` as
`f32` pairs — that's the reference orbit the GPU perturbs from.
`f32` pairs — that's the reference orbit the GPU perturbs from. At
precision ≤ `F64_MAX_PRECISION` (80 bits, i.e. shallow views) it takes a
plain-`f64` fast path (`compute_reference_f64`), so each kind's formula
exists twice in this file (f64 + `FBig`) and both must stay in sync;
`f64_fast_path_matches_big` checks they agree. Requests are made with 1.5×
iteration headroom (`reference_iterations` in `app.rs`), so auto-iterations
creeping up during a zoom doesn't recompute the orbit every frame.
- `src/shaders/*.wgsl` — none of these are standalone WGSL modules; WGSL has
no `#include`, so each is compiled by concatenating plain-text fragments
with `concat!`/`include_str!` at the `create_shader_module` call site (see
@@ -98,7 +104,15 @@ pixel is a handful of `f32` complex multiplies.
Because there's no namespacing, a definition must live in exactly one file
among those concatenated together for a given shader — don't redefine a
`common.wgsl`/`iterate_uniforms.wgsl` symbol locally.
- `src/shaders/mandelbrot.wgsl` — the perturbation fragment shader.
- `src/shaders/mandelbrot.wgsl` — the perturbation fragment shader. It is
**specialized per pipeline** through WGSL `override` constants (`KIND`,
`IS_JULIA`, `DE`), so the per-iteration kind/Julia/DE branches fold away at
pipeline creation. Read those constants in the shader, never `u.kind` /
`u.is_julia` / `u.de_coloring` (they're still uploaded for layout reasons).
`renderer.rs` builds one pipeline set per `PipelineKey` lazily on first
use, and `tests/shader_valid.rs` compiles every kind × Julia × DE variant to
SPIR-V. So a new kind needs no pipeline-list change, only its `KIND_*`
constant. `buddhabrot.wgsl` does the same with its own `override KIND`.
`advance_delta(z, e)` is the per-kind delta step (`z` = reference point,
`e` = current delta); the caller adds `step_add` (= `dc`) afterward — this
relies on `c` being additive in every current kind's formula (a kind where
@@ -111,7 +125,10 @@ pixel is a handful of `f32` complex multiplies.
matching `FractalKind` variant's discriminant exactly.
- `src/fractal/renderer.rs` — `FractalRenderer` (wgpu pipelines, uniform +
storage buffers, bind groups), `Uniforms` (repr(C) layout that must match
the WGSL `Uniforms` struct field-for-field, including padding), and
the WGSL `Uniforms` struct field-for-field, including padding; it includes
CPU-precomputed data: `cm_coef`, the Complex Multibrot binomial
coefficients from `app.rs::complex_binomials`, and `light_count` for the
packed `GpuLight` buffer from `lights.rs::gpu_lights`), and
`FractalCallback` (the `egui_wgpu::CallbackTrait` impl: `prepare()` uploads
changed buffers and decides whether to re-run the iterate pass, the cheap
colourise pass, or just blit the cached texture). Also `ExportRender`, a
@@ -168,7 +185,18 @@ histogram buffer, tone-mapped by a fragment pass every frame. Its own
The interactive path splits iteration (expensive, perturbation) from
colourising (cheap, palette remap) into separate offscreen textures, so
palette/color-scale/offset tweaks skip re-iteration entirely (`geom_differs`
vs `color_differs` in `renderer.rs` decide which pass reruns). While the user
is actively panning/zooming, the app renders downscaled with AA off
(`INTERACT_DOWNSCALE`) and snaps back to full resolution once input settles
(`INTERACT_SETTLE`).
vs `color_differs` in `renderer.rs` decide which pass reruns). A frame where
neither differs uploads and renders nothing and only blits. So any new
uniform field must go into one of those two functions (or the lights
comparison), or changing it won't redraw.
AA is **adaptive** on the interactive path. `fs_data` always iterates 1
sample per pixel. When AA is on, `fs_refine` reads that texture and runs the
2×2 grid only on pixels whose 4-neighbours differ (interior/exterior edge, or
`ci`/DE beyond `AA_CI_EPS`/`AA_DE_EPS`), copying the rest. Colourise then
reads the refined texture. PNG export (`fs_color`) still supersamples every
pixel.
While the user is actively panning/zooming, the app renders downscaled with
AA off (`INTERACT_DOWNSCALE`) and snaps back to full resolution once input
settles (`INTERACT_SETTLE`).
+1 -1
View File
@@ -41,4 +41,4 @@ opt-level = 1
opt-level = 3
[dev-dependencies]
naga = { version = "30", features = ["wgsl-in"] }
naga = { version = "30", features = ["wgsl-in", "spv-out"] }
+49 -7
View File
@@ -15,7 +15,7 @@ use crate::fractal::{
FractalKind, FractalRenderer, MAX_REF_POINTS, ShareState, Uniforms, compute_reference,
compute_set_reference,
};
use crate::lights::Light;
use crate::lights::{Light, gpu_lights};
use crate::view::parse_half_height_spec;
use crate::view::parse_re_im_spec;
use crate::view::{
@@ -757,6 +757,9 @@ impl FractalApp {
ViewState::with_center(big_from_f64(cr, 53), big_from_f64(ci, 53), hh)
}
/// The request key for the current state. Its `iter` is the reference
/// length to compute, which carries headroom over `max_iterations` (see
/// [`reference_iterations`]).
fn current_key(&self) -> RequestKey {
RequestKey {
center_re: self.view.center_re.clone(),
@@ -766,7 +769,7 @@ impl FractalApp {
julia_c: self.julia_c,
phoenix_p: self.phoenix_p,
lambda_l: self.lambda_l,
iter: self.max_iterations,
iter: reference_iterations(self.max_iterations),
kind: self.kind,
power: self.power,
complex_power: self.complex_power,
@@ -792,7 +795,12 @@ impl FractalApp {
|| key.julia_c != self.julia_c
|| key.phoenix_p != self.phoenix_p
|| key.lambda_l != self.lambda_l
|| key.iter != self.max_iterations
// The reference is computed with headroom, so it keeps serving
// while auto-iterations creep up during a zoom (the shader clamps
// to `max_iterations`); only recompute once it's too short, or
// far longer than needed.
|| self.max_iterations > key.iter
|| self.max_iterations.saturating_mul(4) < key.iter
|| key.kind != self.kind
|| key.power != self.power
|| key.complex_power != self.complex_power
@@ -935,8 +943,10 @@ impl FractalApp {
self.max_iterations = self.auto_iteration_count();
}
let mut key = self.current_key();
// One-shot render: no later frames for iteration headroom to serve.
key.iter = self.max_iterations.min(MAX_REF_POINTS as u32 - 1);
let precision = self.view.precision_bits();
let max_iter = key.iter.min(MAX_REF_POINTS as u32 - 1);
let max_iter = key.iter;
// Lambda in Set mode has a static fractal centered at origin.
if key.kind == FractalKind::Lambda && !key.julia {
@@ -1014,8 +1024,9 @@ impl FractalApp {
.inverse()
.to_cols_array(),
screen_dim: self.screen_dim,
light_count: gpu_lights(&self.lights).1,
cm_coef: complex_binomials(self.complex_power),
_pad: [0; _],
_pad2: [0; _],
_pad3: [0; _],
}
}
@@ -1086,7 +1097,7 @@ impl FractalApp {
self.status = Some("export unavailable".into());
return;
};
renderer.export_handles()
renderer.export_handles(&device, &uniforms)
};
let reference = Arc::clone(&self.reference);
let lights = self.lights.clone();
@@ -2387,7 +2398,7 @@ impl FractalApp {
rect,
FractalCallback {
uniforms,
lights: self.lights.clone(),
lights: gpu_lights(&self.lights).0,
reference: Arc::clone(&self.reference),
generation: self.generation,
size_px,
@@ -2441,6 +2452,37 @@ impl eframe::App for FractalApp {
}
}
/// Reference-orbit length to request for `max_iterations`: 1.5× headroom
/// (capped at the GPU buffer size). Auto-iterations grows with every zoom
/// frame, and without headroom each tiny increase re-ran the whole
/// high-precision orbit (plus a re-upload) on every frame of a zoom.
fn reference_iterations(max_iterations: u32) -> u32 {
let cap = MAX_REF_POINTS as u32 - 1;
(max_iterations.saturating_add(max_iterations / 2)).min(cap)
}
/// Complex binomial coefficients `C(p, k)` for k = 1..16, packed two per row
/// (odd k in `[0..2]`, even k in `[2..4]`) for `Uniforms::cm_coef`: the
/// Complex Multibrot delta series' coefficients, which only depend on the
/// power, so the shader doesn't rebuild them (with a complex division per
/// term) on every iteration of every pixel. Built up in f64 via
/// `C(p,k) = C(p,k-1) * (p - (k-1)) / k`.
fn complex_binomials(p: (f64, f64)) -> [[f32; 4]; 8] {
let mut out = [[0.0f32; 4]; 8];
let (mut cr, mut ci) = (1.0f64, 0.0f64); // C(p, 0)
for k in 1..=16usize {
// (cr + i ci) * ((p.0 - (k-1)) + i p.1) / k
let (ar, ai) = (p.0 - (k - 1) as f64, p.1);
let kf = k as f64;
(cr, ci) = ((cr * ar - ci * ai) / kf, (cr * ai + ci * ar) / kf);
let row = &mut out[(k - 1) / 2];
let col = if k % 2 == 1 { 0 } else { 2 };
row[col] = cr as f32;
row[col + 1] = ci as f32;
}
out
}
/// Update an export's progress (phase label + fraction).
fn set_progress(shared: &Arc<Mutex<ExportShared>>, phase: &'static str, fraction: f32) {
let mut s = shared.lock().unwrap();
+32 -11
View File
@@ -3,6 +3,8 @@
//! to colour by a fragment pass. See `shaders/buddhabrot.wgsl` for the "why"
//! this is a separate pipeline from the escape-time perturbation renderer.
use std::collections::HashMap;
use eframe::egui_wgpu::{self, wgpu};
/// Random samples dispatched per accumulating frame. Chosen so a frame stays
@@ -101,7 +103,12 @@ struct Histogram {
}
pub struct BuddhabrotRenderer {
compute_pipeline: wgpu::ComputePipeline,
shader: wgpu::ShaderModule,
compute_pipeline_layout: wgpu::PipelineLayout,
/// Accumulation pipelines, specialized per fractal kind (the shader's
/// `override KIND`, so `advance()` has no per-step kind branches) and
/// built lazily on first use.
compute_pipelines: HashMap<u32, wgpu::ComputePipeline>,
compute_bind_group_layout: wgpu::BindGroupLayout,
tonemap_pipeline: wgpu::RenderPipeline,
tonemap_bind_group_layout: wgpu::BindGroupLayout,
@@ -168,14 +175,6 @@ impl BuddhabrotRenderer {
bind_group_layouts: &[Some(&compute_bind_group_layout)],
immediate_size: 0,
});
let compute_pipeline = device.create_compute_pipeline(&wgpu::ComputePipelineDescriptor {
label: Some("buddhabrot compute pipeline"),
layout: Some(&compute_pipeline_layout),
module: &shader,
entry_point: Some("cs_main"),
compilation_options: Default::default(),
cache: None,
});
let tonemap_bind_group_layout =
device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
@@ -236,7 +235,9 @@ impl BuddhabrotRenderer {
});
Self {
compute_pipeline,
shader,
compute_pipeline_layout,
compute_pipelines: HashMap::new(),
compute_bind_group_layout,
tonemap_pipeline,
tonemap_bind_group_layout,
@@ -248,6 +249,23 @@ impl BuddhabrotRenderer {
}
}
/// The accumulation pipeline for `kind`, built on first use.
fn compute_pipeline(&mut self, device: &wgpu::Device, kind: u32) -> &wgpu::ComputePipeline {
self.compute_pipelines.entry(kind).or_insert_with(|| {
device.create_compute_pipeline(&wgpu::ComputePipelineDescriptor {
label: Some("buddhabrot compute pipeline"),
layout: Some(&self.compute_pipeline_layout),
module: &self.shader,
entry_point: Some("cs_main"),
compilation_options: wgpu::PipelineCompilationOptions {
constants: &[("KIND", kind as f64)],
..Default::default()
},
cache: None,
})
})
}
/// Ensure the histogram buffer exists at `width`×`height`, recreating (and
/// resetting accumulation) on a size change.
fn ensure_histogram(&mut self, device: &wgpu::Device, width: u32, height: u32) {
@@ -334,6 +352,9 @@ impl egui_wgpu::CallbackTrait for BuddhabrotCallback {
let width = self.size_px[0].max(1);
let height = self.size_px[1].max(1);
renderer.ensure_histogram(device, width, height);
let pipeline = renderer
.compute_pipeline(device, self.uniforms.kind)
.clone();
let content = ContentKey::from(&self.uniforms);
let content_changed = renderer.last_content != Some(content);
@@ -365,7 +386,7 @@ impl egui_wgpu::CallbackTrait for BuddhabrotCallback {
label: Some("buddhabrot accumulate pass"),
timestamp_writes: None,
});
pass.set_pipeline(&renderer.compute_pipeline);
pass.set_pipeline(&pipeline);
pass.set_bind_group(0, &histogram.compute_bind_group, &[]);
let workgroups = SAMPLES_PER_DISPATCH.div_ceil(WORKGROUP_SIZE);
pass.dispatch_workgroups(workgroups, 1, 1);
+178 -3
View File
@@ -18,6 +18,22 @@ use crate::view::{Big, big_from_f64};
/// bailout before the stored orbit runs out.
const REFERENCE_ESCAPE_SQ: f64 = 1.0e10;
/// Up to this working precision (bits) the orbit is iterated in plain `f64`
/// instead of `FBig` — orders of magnitude faster, which matters most on the
/// web (where the reference is computed inline on the UI thread).
///
/// `precision_for` asks for `zoom_bits + 48` guard bits, but the GPU only
/// consumes the orbit as f32 deltas, so two things actually matter:
/// * Each f64 step's rounding (~1e-16 relative) acts like a tiny local error
/// in the pixel orbits too (perturbation reproduces whatever orbit it's
/// given), far below the f32 delta noise — the orbit only has to be a
/// consistent orbit, not the exact one.
/// * The reference center gets rounded to f64 (<= ~2.2e-16 absolute for
/// |c| <= 2), which shifts the image. At 80 bits (zoom_bits <= 32, i.e.
/// half-height >= ~2.3e-10) a pixel is >= ~5e-13 wide, so that shift stays
/// below 0.1% of a pixel.
const F64_MAX_PRECISION: usize = 80;
/// Compute the reference orbit `Z_0..Z_{len-1}` where `Z_0 = z0` and
/// `Z_{n+1} = f(Z_n, c)` for the given `kind` (and `power`, for Multibrot), up
/// to `max_iter` steps at `precision` bits. Each entry is `[re, im]` in f32.
@@ -34,6 +50,122 @@ pub fn compute_reference(
phoenix_p: (f64, f64),
lambda_l: (f64, f64),
complex_power: (f64, f64),
) -> Vec<[f32; 2]> {
if precision <= F64_MAX_PRECISION {
return compute_reference_f64(
(z0_re.to_f64().value(), z0_im.to_f64().value()),
(c_re.to_f64().value(), c_im.to_f64().value()),
max_iter,
kind,
power,
phoenix_p,
lambda_l,
complex_power,
);
}
compute_reference_big(
z0_re,
z0_im,
c_re,
c_im,
max_iter,
precision,
kind,
power,
phoenix_p,
lambda_l,
complex_power,
)
}
/// [`compute_reference`]'s fast path for shallow views (see
/// [`F64_MAX_PRECISION`]): the same per-kind formulas in plain `f64`.
#[allow(clippy::too_many_arguments)]
fn compute_reference_f64(
z0: (f64, f64),
c: (f64, f64),
max_iter: u32,
kind: FractalKind,
power: u32,
phoenix_p: (f64, f64),
lambda_l: (f64, f64),
complex_power: (f64, f64),
) -> Vec<[f32; 2]> {
let (cr, ci) = c;
let (mut zr, mut zi) = z0;
// Previous iterate, for the Phoenix two-term recurrence (Y_{-1} = 0).
let (mut zr_prev, mut zi_prev) = (0.0f64, 0.0f64);
let (pr, pi) = phoenix_p;
let (lr, li) = lambda_l;
let mut points: Vec<[f32; 2]> = Vec::with_capacity(max_iter as usize + 1);
for _ in 0..=max_iter {
points.push([zr as f32, zi as f32]);
if zr * zr + zi * zi > REFERENCE_ESCAPE_SQ {
break;
}
let (new_zr, new_zi) = match kind {
FractalKind::Mandelbrot => ((zr + zi) * (zr - zi) + cr, 2.0 * zr * zi + ci),
FractalKind::BurningShip => (zr * zr - zi * zi + cr, (2.0 * zr * zi).abs() + ci),
FractalKind::Tricorn => (zr * zr - zi * zi + cr, ci - 2.0 * zr * zi),
FractalKind::Multibrot => {
let (mut rr, mut ri) = (1.0f64, 0.0f64);
for _ in 0..power.max(2) {
(rr, ri) = (rr * zr - ri * zi, rr * zi + ri * zr);
}
(rr + cr, ri + ci)
}
FractalKind::Celtic => ((zr * zr - zi * zi).abs() + cr, 2.0 * zr * zi + ci),
FractalKind::Perpendicular => (zr * zr - zi * zi + cr, ci - 2.0 * zr * zi.abs()),
FractalKind::Buffalo => ((zr * zr - zi * zi).abs() + cr, ci - (2.0 * zr * zi).abs()),
FractalKind::Phoenix => (
zr * zr - zi * zi + cr + (pr * zr_prev - pi * zi_prev),
2.0 * zr * zi + ci + (pr * zi_prev + pi * zr_prev),
),
FractalKind::Lambda => {
// λ·z(1 - z).
let (re2, im2) = (1.0 - zr, -zi);
let (lzr, lzi) = (lr * zr - li * zi, lr * zi + li * zr);
(lzr * re2 - lzi * im2, re2 * lzi + lzr * im2)
}
FractalKind::ComplexMultibrot => {
let (pr, pi) = complex_pow_complex_f64(zr, zi, complex_power.0, complex_power.1);
(pr + cr, pi + ci)
}
};
(zr_prev, zi_prev) = (zr, zi);
(zr, zi) = (new_zr, new_zi);
}
points
}
/// `f64` twin of [`complex_pow_complex`] (principal branch, `0^p = 0`).
fn complex_pow_complex_f64(zr: f64, zi: f64, pr: f64, pi: f64) -> (f64, f64) {
if zr == 0.0 && zi == 0.0 {
return (0.0, 0.0);
}
let ln_r = 0.5 * (zr * zr + zi * zi).ln();
let theta = zi.atan2(zr);
let mag = (pr * ln_r - pi * theta).exp();
let (sin_a, cos_a) = (pr * theta + pi * ln_r).sin_cos();
(mag * cos_a, mag * sin_a)
}
/// [`compute_reference`] at arbitrary precision (`FBig`), for deep views.
#[allow(clippy::too_many_arguments)]
fn compute_reference_big(
z0_re: &Big,
z0_im: &Big,
c_re: &Big,
c_im: &Big,
max_iter: u32,
precision: usize,
kind: FractalKind,
power: u32,
phoenix_p: (f64, f64),
lambda_l: (f64, f64),
complex_power: (f64, f64),
) -> Vec<[f32; 2]> {
let cr = c_re.clone().with_precision(precision).value();
let ci = c_im.clone().with_precision(precision).value();
@@ -67,8 +199,9 @@ pub fn compute_reference(
let (new_zr, new_zi) = match kind {
FractalKind::Mandelbrot => {
// Z^2 = (zr^2 - zi^2) + (2 zr zi) i.
let re = &zr.sqr() - &zi.sqr() + &cr;
// Z^2 = (zr^2 - zi^2) + (2 zr zi) i, with zr^2 - zi^2 as
// (zr + zi)(zr - zi): one multiply instead of two squares.
let re = (&zr + &zi) * (&zr - &zi) + &cr;
let im = ((&zr * &zi) << 1) + &ci; // << 1 is exact ×2 in base 2
(re, im)
}
@@ -97,7 +230,11 @@ pub fn compute_reference(
FractalKind::Perpendicular => {
// (x^2 - y^2) - 2·x·|y| i: abs the imaginary input.
let re = &zr.sqr() - &zi.sqr() + &cr;
let im = &ci - ((&zr * &big_abs(zi.clone())) << 1);
let im = if zi.to_f64().value() < 0.0 {
&ci + ((&zr * &zi) << 1)
} else {
&ci - ((&zr * &zi) << 1)
};
(re, im)
}
FractalKind::Buffalo => {
@@ -263,6 +400,44 @@ mod tests {
}
}
/// The f64 fast path (shallow views) must produce the same orbit as the
/// arbitrary-precision path, for every kind, in both planes.
#[test]
fn f64_fast_path_matches_big() {
let bits_fast = F64_MAX_PRECISION;
let bits_big = F64_MAX_PRECISION + 64;
for kind in FractalKind::ALL {
for julia in [false, true] {
let run = |bits: usize| {
let (a, b) = (big_from_f64(-0.3, bits), big_from_f64(0.2, bits));
let (jr, ji) = (big_from_f64(-0.4, bits), big_from_f64(0.55, bits));
let args = (60, bits, kind, 3, (0.1, -0.2), (0.9, 0.3), (2.3, 0.4));
if julia {
compute_reference(
&a, &b, &jr, &ji, args.0, args.1, args.2, args.3, args.4, args.5,
args.6,
)
} else {
compute_set_reference(
&a, &b, args.0, args.1, args.2, args.3, args.4, args.5, args.6,
)
}
};
let (fast, big) = (run(bits_fast), run(bits_big));
assert_eq!(fast.len(), big.len(), "{kind:?} julia={julia}: length");
for (i, (f, b)) in fast.iter().zip(&big).enumerate() {
for k in 0..2 {
let tol = 1e-5 * (1.0 + b[k].abs());
assert!(
(f[k] - b[k]).abs() <= tol,
"{kind:?} julia={julia}: point {i} {f:?} vs {b:?}"
);
}
}
}
}
}
/// A point inside the main cardioid never escapes: full-length orbit.
#[test]
fn interior_orbit_runs_full_length() {
+332 -145
View File
@@ -9,11 +9,12 @@
//! not a full fractal recompute. The fragment shader iterates each pixel as an
//! f32 perturbation delta from the reference orbit stored in `ref_buffer`.
use std::collections::HashMap;
use std::sync::Arc;
use eframe::egui_wgpu::{self, wgpu};
use crate::lights::{Light, MAX_LIGHT_COUNT};
use crate::lights::{GpuLight, Light, MAX_LIGHT_COUNT, gpu_lights};
/// Maximum reference-orbit length (points) the storage buffer can hold. Also
/// bounds the iteration count. 128k points * 8 bytes = 1 MiB.
@@ -27,7 +28,7 @@ pub const MAX_REF_POINTS: usize = 1 << 17;
const DATA_FORMAT: wgpu::TextureFormat = wgpu::TextureFormat::Rgba32Float;
/// True when the two uniforms differ in any field the iteration pass depends on
/// (i.e. anything except the palette / colour scale / offset).
/// (i.e. anything except the palette / colour scale / offset / camera).
fn geom_differs(a: &Uniforms, b: &Uniforms) -> bool {
a.span != b.span
|| a.max_iter != b.max_iter
@@ -40,17 +41,106 @@ fn geom_differs(a: &Uniforms, b: &Uniforms) -> bool {
|| a.complex_power != b.complex_power
|| a.dc_offset != b.dc_offset
|| a.phoenix_p != b.phoenix_p
|| a.lambda_l != b.lambda_l
|| a.de_coloring != b.de_coloring
// The iterate pass's DE clamp (`max_de`) depends on whether any
// shadow-style mode is on.
|| (a.rendering_mode != 0) != (b.rendering_mode != 0)
}
/// True when the two uniforms differ in a colour-only field (remappable by the
/// cheap colourise pass without re-iterating).
/// True when the two uniforms differ in a field only the colourise pass reads
/// (remappable without re-iterating): palette / colour scale / offset, the
/// shadow style and light count, and the 3D raymarch camera.
fn color_differs(a: &Uniforms, b: &Uniforms) -> bool {
a.color_offset != b.color_offset
|| a.color_scale != b.color_scale
|| a.palette_id != b.palette_id
|| a.shadow_palette_id != b.shadow_palette_id
|| a.rendering_mode != b.rendering_mode
|| a.light_count != b.light_count
|| a.camera_direction != b.camera_direction
|| a.camera_inv_proj != b.camera_inv_proj
|| a.screen_dim != b.screen_dim
}
/// Specialization of the iteration shader (`mandelbrot.wgsl`'s `override`
/// constants). Everything the per-iteration loop branches on is baked into
/// the pipeline instead of tested per step; one pipeline set per key is built
/// lazily on first use (a new `FractalKind` needs nothing here).
#[derive(Copy, Clone, PartialEq, Eq, Hash, Debug)]
pub struct PipelineKey {
kind: u32,
julia: bool,
de: bool,
}
impl PipelineKey {
pub fn from_uniforms(u: &Uniforms) -> Self {
Self {
kind: u.kind,
julia: u.is_julia != 0,
de: u.de_coloring != 0,
}
}
fn constants(&self) -> [(&'static str, f64); 3] {
[
("KIND", self.kind as f64),
("IS_JULIA", self.julia as u32 as f64),
("DE", self.de as u32 as f64),
]
}
}
/// Build a fullscreen-triangle render pipeline (`vs_main` + `fs_entry`)
/// writing a single `format` target, with `constants` for the shader's
/// `override`s.
fn fullscreen_pipeline(
device: &wgpu::Device,
label: &str,
module: &wgpu::ShaderModule,
layout: &wgpu::PipelineLayout,
fs_entry: &str,
format: wgpu::TextureFormat,
constants: &[(&str, f64)],
) -> wgpu::RenderPipeline {
let compilation_options = wgpu::PipelineCompilationOptions {
constants,
..Default::default()
};
device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
label: Some(label),
layout: Some(layout),
vertex: wgpu::VertexState {
module,
entry_point: Some("vs_main"),
buffers: &[],
compilation_options: compilation_options.clone(),
},
fragment: Some(wgpu::FragmentState {
module,
entry_point: Some(fs_entry),
targets: &[Some(wgpu::ColorTargetState {
format,
blend: None,
write_mask: wgpu::ColorWrites::ALL,
})],
compilation_options,
}),
primitive: wgpu::PrimitiveState::default(),
depth_stencil: None,
multisample: wgpu::MultisampleState::default(),
multiview_mask: None,
cache: None,
})
}
/// The interactive iteration pipelines for one [`PipelineKey`].
struct IteratePipelines {
/// 1-spp perturbation iterate → data texture (`fs_data`).
iterate: wgpu::RenderPipeline,
/// Adaptive AA: data texture → AA data texture (`fs_refine`).
refine: wgpu::RenderPipeline,
}
/// GPU-side view + coloring parameters. Layout must match `Uniforms` in the
@@ -97,25 +187,36 @@ pub struct Uniforms {
pub rendering_mode: u32,
// camera direction vector
pub camera_direction: [f32; 3],
pub _pad2: [u32; 1],
/// Number of live entries in the lights buffer (see `gpu_lights`).
pub light_count: u32,
/// Inverse of the camera's view-projection matrix (column-major), for
/// reconstructing a world-space ray origin per pixel in the raymarcher.
pub camera_inv_proj: [f32; 16],
/// Screen dimension
pub screen_dim: [f32; 2],
pub _pad3: [u32; 2],
/// Complex binomial coefficients `C(complex_power, k)`, k = 1..16, two per
/// row (odd k in `[0..2]`, even k in `[2..4]`), for the Complex Multibrot
/// delta series. Derived from `complex_power` alone.
pub cm_coef: [[f32; 4]; 8],
}
/// Offscreen textures for the two-pass render, recreated whenever the widget's
/// pixel size changes:
/// * `data_view` — the iteration pass's output (see [`DATA_FORMAT`]).
/// * `data_view` — the 1-spp iteration pass's output (see [`DATA_FORMAT`]).
/// * `data_aa_view` — the adaptive-AA refine pass's output (only when AA is on).
/// * `color_view` — the colourise pass's output; the blit source.
/// plus the bind groups that read them.
struct CacheTarget {
data_view: wgpu::TextureView,
data_aa_view: wgpu::TextureView,
color_view: wgpu::TextureView,
/// Colourise pass input: uniforms + the data texture.
/// Refine pass input (group 1): the 1-spp data texture.
refine_bind_group: wgpu::BindGroup,
/// Colourise pass input: uniforms + the 1-spp data texture.
colorize_bind_group: wgpu::BindGroup,
/// Colourise pass input when AA is on: uniforms + the refined texture.
colorize_aa_bind_group: wgpu::BindGroup,
/// Blit pass input: the colour texture + sampler.
blit_bind_group: wgpu::BindGroup,
width: u32,
@@ -135,15 +236,21 @@ struct IterState {
/// inputs (and size) match and iteration did not re-run, colourise is skipped.
struct ColorState {
uniforms: Uniforms,
lights: [GpuLight; MAX_LIGHT_COUNT],
width: u32,
height: u32,
}
pub struct FractalRenderer {
/// Iteration pass: perturbation iterate → data texture (`fs_data`).
iterate_pipeline: wgpu::RenderPipeline,
/// Combined iterate + colour in one pass (`fs_color`), used only by export.
export_pipeline: wgpu::RenderPipeline,
/// `mandelbrot.wgsl`, specialized per [`PipelineKey`] at pipeline creation.
shader: wgpu::ShaderModule,
/// Layout of the iterate + export pipelines (group 0 only).
pipeline_layout: wgpu::PipelineLayout,
/// Layout of the refine pipeline (group 0 + the 1-spp texture in group 1).
refine_pipeline_layout: wgpu::PipelineLayout,
refine_bind_group_layout: wgpu::BindGroupLayout,
/// Lazily built interactive pipelines, per shader specialization.
pipelines: HashMap<PipelineKey, IteratePipelines>,
bind_group_layout: wgpu::BindGroupLayout,
uniform_buffer: wgpu::Buffer,
ref_buffer: wgpu::Buffer,
@@ -152,6 +259,8 @@ pub struct FractalRenderer {
target_format: wgpu::TextureFormat,
/// Generation of the reference orbit currently uploaded to `ref_buffer`.
uploaded_generation: u64,
/// Contents of `lights_buffer`, so it's only re-uploaded on change.
uploaded_lights: Option<[GpuLight; MAX_LIGHT_COUNT]>,
/// Colourise pass: data texture → colour texture (palette mapping).
colorize_pipeline: wgpu::RenderPipeline,
@@ -199,7 +308,7 @@ impl FractalRenderer {
let lights_buffer = device.create_buffer(&wgpu::BufferDescriptor {
label: Some("lights parameters"),
size: (MAX_LIGHT_COUNT * std::mem::size_of::<Light>()) as u64,
size: std::mem::size_of::<[GpuLight; MAX_LIGHT_COUNT]>() as u64,
usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
mapped_at_creation: false,
});
@@ -268,58 +377,28 @@ impl FractalRenderer {
immediate_size: 0,
});
// Iteration pass: perturbation iterate → data texture (color-independent).
let iterate_pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
label: Some("fractal iterate pipeline"),
layout: Some(&pipeline_layout),
vertex: wgpu::VertexState {
module: &shader,
entry_point: Some("vs_main"),
buffers: &[],
compilation_options: Default::default(),
// The iterate/refine/export pipelines are specialized per fractal
// kind (see `PipelineKey`) and built lazily; only their layouts are
// fixed. Refine additionally reads the 1-spp data texture (group 1).
let refine_bind_group_layout =
device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
label: Some("refine bind group layout"),
entries: &[wgpu::BindGroupLayoutEntry {
binding: 0,
visibility: wgpu::ShaderStages::FRAGMENT,
ty: wgpu::BindingType::Texture {
sample_type: wgpu::TextureSampleType::Float { filterable: false },
view_dimension: wgpu::TextureViewDimension::D2,
multisampled: false,
},
fragment: Some(wgpu::FragmentState {
module: &shader,
entry_point: Some("fs_data"),
targets: &[Some(wgpu::ColorTargetState {
format: DATA_FORMAT,
blend: None,
write_mask: wgpu::ColorWrites::ALL,
})],
compilation_options: Default::default(),
}),
primitive: wgpu::PrimitiveState::default(),
depth_stencil: None,
multisample: wgpu::MultisampleState::default(),
multiview_mask: None,
cache: None,
count: None,
}],
});
// Combined iterate + colour in one pass — for PNG export only.
let export_pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
label: Some("fractal export pipeline"),
layout: Some(&pipeline_layout),
vertex: wgpu::VertexState {
module: &shader,
entry_point: Some("vs_main"),
buffers: &[],
compilation_options: Default::default(),
},
fragment: Some(wgpu::FragmentState {
module: &shader,
entry_point: Some("fs_color"),
targets: &[Some(wgpu::ColorTargetState {
format: target_format,
blend: None,
write_mask: wgpu::ColorWrites::ALL,
})],
compilation_options: Default::default(),
}),
primitive: wgpu::PrimitiveState::default(),
depth_stencil: None,
multisample: wgpu::MultisampleState::default(),
multiview_mask: None,
cache: None,
let refine_pipeline_layout =
device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
label: Some("refine pipeline layout"),
bind_group_layouts: &[Some(&bind_group_layout), Some(&refine_bind_group_layout)],
immediate_size: 0,
});
// Colourise pass: data texture + colour uniforms → colour texture.
@@ -478,8 +557,11 @@ impl FractalRenderer {
});
Self {
iterate_pipeline,
export_pipeline,
shader,
pipeline_layout,
refine_pipeline_layout,
refine_bind_group_layout,
pipelines: HashMap::new(),
bind_group_layout,
uniform_buffer,
ref_buffer,
@@ -487,6 +569,7 @@ impl FractalRenderer {
bind_group,
target_format,
uploaded_generation: u64::MAX,
uploaded_lights: None,
colorize_pipeline,
colorize_bind_group_layout,
blit_pipeline,
@@ -527,6 +610,19 @@ impl FractalRenderer {
});
let data_view = data_texture.create_view(&wgpu::TextureViewDescriptor::default());
// Adaptive-AA output: same format, written by the refine pass.
let data_aa_texture = device.create_texture(&wgpu::TextureDescriptor {
label: Some("fractal data (AA)"),
size: extent,
mip_level_count: 1,
sample_count: 1,
dimension: wgpu::TextureDimension::D2,
format: DATA_FORMAT,
usage: wgpu::TextureUsages::RENDER_ATTACHMENT | wgpu::TextureUsages::TEXTURE_BINDING,
view_formats: &[],
});
let data_aa_view = data_aa_texture.create_view(&wgpu::TextureViewDescriptor::default());
// Colour texture (colourise output; blit source).
let color_texture = device.create_texture(&wgpu::TextureDescriptor {
label: Some("fractal color cache"),
@@ -540,7 +636,8 @@ impl FractalRenderer {
});
let color_view = color_texture.create_view(&wgpu::TextureViewDescriptor::default());
let colorize_bind_group = device.create_bind_group(&wgpu::BindGroupDescriptor {
let colorize_bind_group_for = |data: &wgpu::TextureView| {
device.create_bind_group(&wgpu::BindGroupDescriptor {
label: Some("colorize bind group"),
layout: &self.colorize_bind_group_layout,
entries: &[
@@ -550,13 +647,25 @@ impl FractalRenderer {
},
wgpu::BindGroupEntry {
binding: 1,
resource: wgpu::BindingResource::TextureView(&data_view),
resource: wgpu::BindingResource::TextureView(data),
},
wgpu::BindGroupEntry {
binding: 2,
resource: self.lights_buffer.as_entire_binding(),
},
],
})
};
let colorize_bind_group = colorize_bind_group_for(&data_view);
let colorize_aa_bind_group = colorize_bind_group_for(&data_aa_view);
let refine_bind_group = device.create_bind_group(&wgpu::BindGroupDescriptor {
label: Some("refine bind group"),
layout: &self.refine_bind_group_layout,
entries: &[wgpu::BindGroupEntry {
binding: 0,
resource: wgpu::BindingResource::TextureView(&data_view),
}],
});
let blit_bind_group = device.create_bind_group(&wgpu::BindGroupDescriptor {
@@ -576,8 +685,11 @@ impl FractalRenderer {
self.cache = Some(CacheTarget {
data_view,
data_aa_view,
color_view,
refine_bind_group,
colorize_bind_group,
colorize_aa_bind_group,
blit_bind_group,
width,
height,
@@ -587,21 +699,57 @@ impl FractalRenderer {
self.colored = None;
}
/// Build (on first use) and cache the interactive pipelines for `key`.
fn ensure_pipelines(&mut self, device: &wgpu::Device, key: PipelineKey) {
if self.pipelines.contains_key(&key) {
return;
}
let constants = key.constants();
let iterate = fullscreen_pipeline(
device,
"fractal iterate pipeline",
&self.shader,
&self.pipeline_layout,
"fs_data",
DATA_FORMAT,
&constants,
);
let refine = fullscreen_pipeline(
device,
"fractal AA refine pipeline",
&self.shader,
&self.refine_pipeline_layout,
"fs_refine",
DATA_FORMAT,
&constants,
);
self.pipelines
.insert(key, IteratePipelines { iterate, refine });
}
/// Handles needed to build a standalone [`ExportRender`] off the UI thread:
/// the (immutable) pipeline and its bind-group layout, plus the target
/// format. Cloned so the caller can drop the render-state lock before use.
/// a combined iterate + colour pipeline (`fs_color`) specialized for
/// `uniforms` (built fresh — exports are rare, and this only needs a read
/// lock on the renderer), its bind-group layout, and the target format.
pub fn export_handles(
&self,
device: &wgpu::Device,
uniforms: &Uniforms,
) -> (
wgpu::RenderPipeline,
wgpu::BindGroupLayout,
wgpu::TextureFormat,
) {
(
self.export_pipeline.clone(),
self.bind_group_layout.clone(),
let pipeline = fullscreen_pipeline(
device,
"fractal export pipeline",
&self.shader,
&self.pipeline_layout,
"fs_color",
self.target_format,
)
&PipelineKey::from_uniforms(uniforms).constants(),
);
(pipeline, self.bind_group_layout.clone(), self.target_format)
}
}
@@ -663,14 +811,12 @@ impl ExportRender {
// (zeroed) for every other coloring mode.
let lights_buffer = device.create_buffer(&wgpu::BufferDescriptor {
label: Some("export lights"),
size: (MAX_LIGHT_COUNT * std::mem::size_of::<Light>()) as u64,
size: std::mem::size_of::<[GpuLight; MAX_LIGHT_COUNT]>() as u64,
usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
mapped_at_creation: false,
});
let mut light_bytes = [0u8; size_of::<Light>() * MAX_LIGHT_COUNT];
let n = lights.len().min(MAX_LIGHT_COUNT);
light_bytes[..n * size_of::<Light>()].copy_from_slice(bytemuck::cast_slice(&lights[..n]));
queue.write_buffer(&lights_buffer, 0, &light_bytes);
let (gpu_lights, _) = gpu_lights(lights);
queue.write_buffer(&lights_buffer, 0, bytemuck::cast_slice(&gpu_lights));
let bind_group = device.create_bind_group(&wgpu::BindGroupDescriptor {
label: Some("export bind group"),
@@ -924,14 +1070,49 @@ pub fn encode_png_with_progress(
out
}
/// Record one fullscreen-triangle pass drawing `pipeline` into `target`
/// (cleared first), with `bind_groups` bound to groups 0, 1, ...
fn data_pass(
encoder: &mut wgpu::CommandEncoder,
label: &str,
target: &wgpu::TextureView,
pipeline: &wgpu::RenderPipeline,
bind_groups: &[&wgpu::BindGroup],
) {
let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
label: Some(label),
color_attachments: &[Some(wgpu::RenderPassColorAttachment {
view: target,
depth_slice: None,
resolve_target: None,
ops: wgpu::Operations {
load: wgpu::LoadOp::Clear(wgpu::Color::BLACK),
store: wgpu::StoreOp::Store,
},
})],
depth_stencil_attachment: None,
timestamp_writes: None,
occlusion_query_set: None,
multiview_mask: None,
});
pass.set_pipeline(pipeline);
for (i, bg) in bind_groups.iter().enumerate() {
pass.set_bind_group(i as u32, *bg, &[]);
}
pass.draw(0..3, 0..1);
}
/// A per-frame paint callback. Carries this frame's uniforms plus a reference to
/// the current reference orbit (cheap `Arc` clone). The orbit is only re-uploaded
/// when its `generation` changes; the expensive iteration pass re-runs only when
/// a geometry input changes, and colour-only changes re-run just the cheap
/// colourise pass (see `prepare`).
/// a geometry input changes, colour-only changes re-run just the cheap
/// colourise pass, and a frame where nothing changed (e.g. a hover repaint)
/// uploads and renders nothing — `paint` just blits the cache (see `prepare`).
pub struct FractalCallback {
pub uniforms: Uniforms,
pub lights: Vec<Light>,
/// Lights buffer contents, from [`gpu_lights`] (its count is in
/// `uniforms.light_count`).
pub lights: [GpuLight; MAX_LIGHT_COUNT],
pub reference: Arc<Vec<[f32; 2]>>,
pub generation: u64,
/// Widget size in physical pixels — the cache texture resolution.
@@ -955,7 +1136,32 @@ impl egui_wgpu::CallbackTrait for FractalCallback {
let height = self.size_px[1].max(1);
renderer.ensure_cache(device, width, height);
if renderer.uploaded_generation != self.generation && !self.reference.is_empty() {
// Iteration (expensive) re-runs only when the geometry inputs change;
// colourise (cheap) re-runs when it did, or when only a colour/camera/
// light input changed — so palette tweaks, colour cycling, and 3D
// camera moves skip the perturbation entirely.
let iter_dirty = renderer.iterated.as_ref().is_none_or(|r| {
r.generation != self.generation
|| r.width != width
|| r.height != height
|| geom_differs(&r.uniforms, &self.uniforms)
});
let color_dirty = iter_dirty
|| renderer.colored.as_ref().is_none_or(|c| {
c.width != width
|| c.height != height
|| c.lights != self.lights
|| color_differs(&c.uniforms, &self.uniforms)
});
if !color_dirty {
return Vec::new(); // cache still valid; paint() just blits it
}
if iter_dirty
&& renderer.uploaded_generation != self.generation
&& !self.reference.is_empty()
{
let count = self.reference.len().min(MAX_REF_POINTS);
queue.write_buffer(
&renderer.ref_buffer,
@@ -965,81 +1171,61 @@ impl egui_wgpu::CallbackTrait for FractalCallback {
renderer.uploaded_generation = self.generation;
}
// Iteration (expensive) re-runs only when the geometry inputs change;
// colourise (cheap) re-runs when it did, or when only a colour changed —
// so palette / colour-scale / offset tweaks (e.g. colour cycling) skip
// the perturbation entirely.
let iter_dirty = renderer.iterated.as_ref().is_none_or(|r| {
r.generation != self.generation
|| r.width != width
|| r.height != height
|| geom_differs(&r.uniforms, &self.uniforms)
});
let color_dirty = iter_dirty
|| renderer.colored.as_ref().is_none_or(|c| {
c.width != width || c.height != height || color_differs(&c.uniforms, &self.uniforms)
})
|| true;
if !color_dirty {
return Vec::new(); // cache still valid; paint() just blits it
}
// Both passes read the uniform buffer; refresh it once.
// Every pass reads the uniform buffer; refresh it once.
queue.write_buffer(
&renderer.uniform_buffer,
0,
bytemuck::bytes_of(&self.uniforms),
);
let mut bytes = [0; size_of::<Light>() * MAX_LIGHT_COUNT];
bytes[..self.lights.len() * size_of::<Light>()]
.copy_from_slice(bytemuck::cast_slice(&self.lights));
queue.write_buffer(&renderer.lights_buffer, 0, &bytes);
if renderer.uploaded_lights.as_ref() != Some(&self.lights) {
queue.write_buffer(
&renderer.lights_buffer,
0,
bytemuck::cast_slice(&self.lights),
);
renderer.uploaded_lights = Some(self.lights);
}
let aa = self.uniforms.aa_level > 1;
if iter_dirty {
renderer.ensure_pipelines(device, PipelineKey::from_uniforms(&self.uniforms));
}
let pipelines = &renderer.pipelines[&PipelineKey::from_uniforms(&self.uniforms)];
if let Some(cache) = &renderer.cache {
if iter_dirty {
// Iteration pass: perturbation iterate → data texture.
let mut pass = egui_encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
label: Some("fractal iterate pass"),
color_attachments: &[Some(wgpu::RenderPassColorAttachment {
view: &cache.data_view,
depth_slice: None,
resolve_target: None,
ops: wgpu::Operations {
load: wgpu::LoadOp::Clear(wgpu::Color::BLACK),
store: wgpu::StoreOp::Store,
},
})],
depth_stencil_attachment: None,
timestamp_writes: None,
occlusion_query_set: None,
multiview_mask: None,
});
pass.set_pipeline(&renderer.iterate_pipeline);
pass.set_bind_group(0, &renderer.bind_group, &[]);
pass.draw(0..3, 0..1);
// Iteration pass: 1-spp perturbation iterate → data texture.
data_pass(
egui_encoder,
"fractal iterate pass",
&cache.data_view,
&pipelines.iterate,
&[&renderer.bind_group],
);
if aa {
// Adaptive AA: supersample only the non-smooth pixels.
data_pass(
egui_encoder,
"fractal AA refine pass",
&cache.data_aa_view,
&pipelines.refine,
&[&renderer.bind_group, &cache.refine_bind_group],
);
}
}
// Colourise pass: data texture → colour texture.
let mut pass = egui_encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
label: Some("fractal colorize pass"),
color_attachments: &[Some(wgpu::RenderPassColorAttachment {
view: &cache.color_view,
depth_slice: None,
resolve_target: None,
ops: wgpu::Operations {
load: wgpu::LoadOp::Clear(wgpu::Color::BLACK),
store: wgpu::StoreOp::Store,
},
})],
depth_stencil_attachment: None,
timestamp_writes: None,
occlusion_query_set: None,
multiview_mask: None,
});
pass.set_pipeline(&renderer.colorize_pipeline);
pass.set_bind_group(0, &cache.colorize_bind_group, &[]);
pass.draw(0..3, 0..1);
let colorize_bind_group = if aa {
&cache.colorize_aa_bind_group
} else {
&cache.colorize_bind_group
};
data_pass(
egui_encoder,
"fractal colorize pass",
&cache.color_view,
&renderer.colorize_pipeline,
&[colorize_bind_group],
);
}
if iter_dirty {
@@ -1052,6 +1238,7 @@ impl egui_wgpu::CallbackTrait for FractalCallback {
}
renderer.colored = Some(ColorState {
uniforms: self.uniforms,
lights: self.lights,
width,
height,
});
+6 -3
View File
@@ -64,9 +64,9 @@ pub fn run(cli: Cli) -> Result<(), String> {
let (device, queue) = pollster::block_on(request_device())?;
let format = wgpu::TextureFormat::Bgra8Unorm;
let renderer = FractalRenderer::new(&device, format);
let (pipeline, bind_group_layout, format) = renderer.export_handles();
let uniforms = app.make_uniforms(width as f64 / height as f64);
let (pipeline, bind_group_layout, format) = renderer.export_handles(&device, &uniforms);
let er = ExportRender::new(
&device,
&queue,
@@ -141,7 +141,10 @@ fn run_animation(
let (device, queue) = pollster::block_on(request_device())?;
let format = wgpu::TextureFormat::Bgra8Unorm;
let renderer = FractalRenderer::new(&device, format);
let (pipeline, bind_group_layout, format) = renderer.export_handles();
// Only the camera animates, so the shader specialization (kind, Julia,
// DE) is the same for every frame.
let (pipeline, bind_group_layout, format) =
renderer.export_handles(&device, &app.make_uniforms(width as f64 / height as f64));
for i in 0..frames {
let raw_t = i as f64 / (frames - 1) as f64;
+32
View File
@@ -60,3 +60,35 @@ impl Light {
.inner
}
}
/// GPU-side light, matching WGSL `Light` in `iterate_uniforms.wgsl`: the unit
/// direction toward the light (precomputed from azimuth/altitude so the
/// shader does no per-pixel trig) plus the packed RGBA colour, whose alpha is
/// the intensity. 16 bytes, so `array<Light, 16>` has a uniform-legal stride.
#[derive(Clone, Copy, PartialEq, Zeroable, Pod, Default)]
#[repr(C)]
pub struct GpuLight {
pub dir: [f32; 3],
pub color: Color32,
}
/// The light buffer's contents: the UI lights with a non-zero colour (the
/// only ones that contribute, and the ones the filmic white point counts),
/// packed to the front, plus how many there are (`Uniforms::light_count`).
pub fn gpu_lights(lights: &[Light]) -> ([GpuLight; MAX_LIGHT_COUNT], u32) {
let mut out = [GpuLight::default(); MAX_LIGHT_COUNT];
let mut n = 0;
for l in lights.iter().filter(|l| l.color != Color32::TRANSPARENT) {
if n == MAX_LIGHT_COUNT {
break;
}
let (sa, ca) = l.altitude.sin_cos();
let (sz, cz) = l.azimuth.sin_cos();
out[n] = GpuLight {
dir: [cz * ca, sz * ca, sa],
color: l.color,
};
n += 1;
}
(out, n as u32)
}
+16 -10
View File
@@ -58,6 +58,12 @@ struct Uniforms {
complex_power: vec2<f32>,
};
// Fractal kind, as a pipeline-overridable constant (set per compute pipeline
// from `u.kind`, see `BuddhabrotRenderer::compute_pipeline`): every kind
// branch in the iteration loop folds away at pipeline creation. Read this,
// never `u.kind`.
override KIND: u32 = 0u;
const PALETTE_NEBULA: u32 = 0u;
const PALETTE_YELLOW: u32 = 1u;
const PALETTE_GRAYSCALE: u32 = 2u;
@@ -95,25 +101,25 @@ fn complex_pow(z: vec2<f32>, p: u32) -> vec2<f32> {
// Must match `FractalKind` in reference.rs (the direct, non-perturbative form
// of the same formulas).
fn advance(z: vec2<f32>, zp: vec2<f32>, c: vec2<f32>) -> vec2<f32> {
if u.kind == KIND_BURNING_SHIP {
if KIND == KIND_BURNING_SHIP {
return vec2<f32>(z.x * z.x - z.y * z.y, 2.0 * abs(z.x * z.y)) + c;
} else if u.kind == KIND_TRICORN {
} else if KIND == KIND_TRICORN {
return vec2<f32>(z.x * z.x - z.y * z.y, -2.0 * z.x * z.y) + c;
} else if u.kind == KIND_MULTIBROT {
} else if KIND == KIND_MULTIBROT {
return complex_pow(z, clamp(u.power, 2u, 8u)) + c;
} else if u.kind == KIND_CELTIC {
} else if KIND == KIND_CELTIC {
return vec2<f32>(abs(z.x * z.x - z.y * z.y), 2.0 * z.x * z.y) + c;
} else if u.kind == KIND_PERPENDICULAR {
} else if KIND == KIND_PERPENDICULAR {
return vec2<f32>(z.x * z.x - z.y * z.y, -2.0 * z.x * abs(z.y)) + c;
} else if u.kind == KIND_BUFFALO {
} else if KIND == KIND_BUFFALO {
return vec2<f32>(abs(z.x * z.x - z.y * z.y), -abs(2.0 * z.x * z.y)) + c;
} else if u.kind == KIND_PHOENIX {
} else if KIND == KIND_PHOENIX {
let sq = vec2<f32>(z.x * z.x - z.y * z.y, 2.0 * z.x * z.y);
return sq + c + cmul(u.phoenix_p, zp);
} else if u.kind == KIND_LAMBDA {
} else if KIND == KIND_LAMBDA {
// l * z * (1 - z); c is unused (see file doc comment above).
return cmul(u.lambda_l, cmul(z, vec2<f32>(1.0 - z.x, -z.y)));
} else if u.kind == KIND_COMPLEX_MULTIBROT {
} else if KIND == KIND_COMPLEX_MULTIBROT {
return cpow(z, u.complex_power) + c;
}
return vec2<f32>(z.x * z.x - z.y * z.y, 2.0 * z.x * z.y) + c; // Mandelbrot
@@ -176,7 +182,7 @@ fn cs_main(@builtin(global_invocation_id) gid: vec3<u32>) {
var c = sample;
var z0 = vec2<f32>(0.0, 0.0);
if u.kind == KIND_LAMBDA {
if KIND == KIND_LAMBDA {
c = vec2<f32>(0.0, 0.0); // unused by the Lambda step
z0 = sample;
}
+30 -22
View File
@@ -20,18 +20,18 @@ fn vs_main(@builtin(vertex_index) idx: u32) -> @builtin(position) vec4<f32> {
}
fn shadow_fragment(pos: vec2<f32>) -> vec4<f32> {
let x = i32(pos.x);
let y = i32(pos.y);
let size = textureDimensions(data_tex);
if textureLoad(data_tex, vec2<i32>(x, y), 0).b != 0. {
let here = textureLoad(data_tex, vec2<i32>(x, y), 0);
if here.b != 0. {
return vec4<f32>(0.1, 0.1, 0.1, 1.0);
} else {
}
// Forward differences, except on the last column/row where x+1 / y+1
// is off the texture: fall back to a backward difference, mirrored
// (h0 + (h0 - h[-1])) so the slope keeps the sign normal_from_heights
// expects — plugging h[-1] in directly would flip the normal there.
let h0 = textureLoad(data_tex, vec2<i32>(x, y), 0).g;
let h0 = here.g;
var h1: f32;
if x + 1 < i32(size.x) {
h1 = textureLoad(data_tex, vec2<i32>(x + 1, y), 0).g;
@@ -46,8 +46,8 @@ fn shadow_fragment(pos: vec2<f32>) -> vec4<f32> {
}
let normal = normal_from_heights(h0, h1, h2);
return vec4<f32>(shadow_color(normal), 1.0);
}
}
@fragment
fn fs_main(@builtin(position) pos: vec4<f32>) -> @location(0) vec4<f32> {
if u.shadow == 2u {
@@ -68,19 +68,25 @@ fn fs_main(@builtin(position) pos: vec4<f32>) -> @location(0) vec4<f32> {
}
}
fn sdf(pos: vec3<f32>) -> f32 {
let aspect_ratio = u.screen_dim.x / u.screen_dim.y;
let size = vec2<f32>(textureDimensions(data_tex));
var texture_pos_f32 = vec2<f32>(pos.x * size.x / aspect_ratio, pos.y * size.y);
var texture_pos = vec2<i32>(i32(texture_pos_f32.x), i32(texture_pos_f32.y));
texture_pos.x = clamp(texture_pos.x, 0, i32(size.x) - 1);
texture_pos.y = clamp(texture_pos.y, 0, i32(size.y) - 1);
// Per-frame constants of the raymarch, computed once per pixel in
// `ray_marching` rather than on each of the up-to-100 `sdf` steps.
struct MarchConsts {
size: vec2<f32>,
// (size.x / aspect_ratio, size.y): world xy -> texel scale.
to_texel: vec2<f32>,
size_i: vec2<i32>,
inv_size_y: f32,
};
let to_texture = max(-min(texture_pos_f32, vec2(0.)), max(texture_pos_f32 - size, vec2(0.)));
let dist_to_texture = length(to_texture) / size.y;
fn sdf(pos: vec3<f32>, k: MarchConsts) -> f32 {
let texture_pos_f32 = pos.xy * k.to_texel;
let texture_pos = clamp(vec2<i32>(texture_pos_f32), vec2<i32>(0, 0), k.size_i - vec2<i32>(1, 1));
let to_texture = max(-min(texture_pos_f32, vec2(0.)), max(texture_pos_f32 - k.size, vec2(0.)));
let dist_to_texture = length(to_texture) * k.inv_size_y;
let px = textureLoad(data_tex, texture_pos, 0);
let de = (px.g / size.y) * 0.5;
let de = (px.g * k.inv_size_y) * 0.5;
// Height is measured toward -z, the side the camera sits on (it looks
// along +z), so the terrain is solid on +z: interior plateau at z = 0,
// exterior sloping away from the camera as `de` grows.
@@ -105,7 +111,11 @@ fn sdf(pos: vec3<f32>) -> f32 {
}
fn ray_marching(pos: vec4<f32>) -> vec4<f32> {
let size = vec2<f32>(textureDimensions(data_tex));
let size_i = vec2<i32>(textureDimensions(data_tex));
let size = vec2<f32>(size_i);
let aspect_ratio = u.screen_dim.x / u.screen_dim.y;
let k = MarchConsts(size, vec2<f32>(size.x / aspect_ratio, size.y), size_i, 1.0 / size.y);
let in_texture = vec2<f32>(
(pos.x / size.x) * 2. - 1.,
(pos.y / size.y) * 2. - 1.,
@@ -123,10 +133,11 @@ fn ray_marching(pos: vec4<f32>) -> vec4<f32> {
let dist_threshold = 0.000001;
while i < 100u {
if length(p - ray_origin) > 3. {
let from_origin = p - ray_origin;
if dot(from_origin, from_origin) > 9. {
break;
}
dist = sdf(p);
dist = sdf(p, k);
if dist < dist_threshold {
break;
}
@@ -135,10 +146,7 @@ fn ray_marching(pos: vec4<f32>) -> vec4<f32> {
}
if dist < dist_threshold {
let aspect_ratio = u.screen_dim.x / u.screen_dim.y;
let size = vec2<f32>(textureDimensions(data_tex));
let texture_pos_f32 = vec2<f32>(p.x * size.x / aspect_ratio, p.y * size.y);
return shadow_fragment(texture_pos_f32);
return shadow_fragment(p.xy * k.to_texel);
}
return vec4<f32>(1., 0., 0., 1.);
}
+23 -22
View File
@@ -35,11 +35,18 @@ struct Uniforms {
shadow: u32,
// camera direction vector
camera_direction: vec3<f32>,
// Number of live entries at the start of `lights` (fills the vec3's tail
// padding slot).
light_count: u32,
// inverse of the camera's view-projection matrix, for reconstructing a
// world-space ray origin per pixel in the raymarcher
camera_inv_proj: mat4x4<f32>,
// Screen dimensions
screen_dim: vec2<f32>
screen_dim: vec2<f32>,
// Complex binomial coefficients C(complex_power, k) for k = 1..16, two per
// vec4 (k odd in .xy, k even in .zw), for the Complex Multibrot delta
// series. Precomputed on the CPU since they only depend on the power.
cm_coef: array<vec4<f32>, 8>,
};
// Smooth cyclic palettes (Inigo Quilez cosine palettes), selected by id.
@@ -74,20 +81,21 @@ fn classic_color(ci: f32, de: f32) -> vec3<f32> {
return palette(u.palette_id, t) * sqrt(de);
}
// A single directional/point light, set by the UI's light list. `color`'s
// alpha channel doubles as intensity (see `shadow_color`'s use of
// `light_color.a`). Each shader that binds a `lights: array<Light, 16>`
// uniform (colorize.wgsl, mandelbrot.wgsl's export shadow path) uses this
// same layout.
// A single directional light, built on the CPU from the UI's light list
// (`GpuLight` in lights.rs): `dir` is the unit direction toward the light
// (precomputed from azimuth/altitude so the shader does no trig), `color` a
// packed RGBA8 whose alpha doubles as intensity. Only the first
// `u.light_count` entries are live, all with a non-zero colour. Each shader
// that binds a `lights: array<Light, 16>` uniform (colorize.wgsl,
// mandelbrot.wgsl's export shadow path) uses this same layout.
struct Light {
azimuth: f32,
altitude: f32,
dir: vec3<f32>,
color: u32,
_pad: u32,
};
// Lambertian term for a unit `light` direction.
fn compute_light(normal: vec3<f32>, light: vec3<f32>) -> vec3<f32> {
return vec3<f32>(max(0., dot(normal, normalize(light))));
return vec3<f32>(max(0., dot(normal, light)));
}
fn uncharted2tonemap(x: vec3<f32>) -> vec3<f32> {
@@ -139,27 +147,20 @@ fn normal_from_heights(h0: f32, h1: f32, h2: f32) -> vec3<f32> {
fn shadow_color(normal: vec3<f32>) -> vec3<f32> {
var color: vec3<f32>;
if u.shadow_palette_id == 0u {
color = compute_light(normal, vec3<f32>(.5, .5, .5)) + vec3<f32>(0.58, 0.85, 1.) * 0.2;
color = compute_light(normal, vec3<f32>(0.57735027, 0.57735027, 0.57735027)) + vec3<f32>(0.58, 0.85, 1.) * 0.2;
color = filmic(color, 2.5);
color = contrast(color, 4., 0.67);
} else if u.shadow_palette_id == 1u {
color = compute_light(normal, vec3<f32>(0., .5, .5)) * vec3<f32>(1., 0.5, 0.5) + compute_light(normal, vec3<f32>(0.5, 0., .5)) * vec3<f32>(0.5, 1., 1.);
color = compute_light(normal, vec3<f32>(0., 0.70710678, 0.70710678)) * vec3<f32>(1., 0.5, 0.5) + compute_light(normal, vec3<f32>(0.70710678, 0., 0.70710678)) * vec3<f32>(0.5, 1., 1.);
color = filmic(color, 4.2);
} else {
color = vec3<f32>(0);
var light_count = 0;
for (var i = 0u; i < 16; i++) {
let light_count = min(u.light_count, 16u);
for (var i = 0u; i < light_count; i++) {
let light_color = unpack4x8unorm(lights[i].color);
if any(light_color != vec4<f32>(0)) {
light_count += 1;
}
color += compute_light(normal, vec3<f32>(
cos(lights[i].azimuth) * cos(lights[i].altitude),
sin(lights[i].azimuth) * cos(lights[i].altitude),
sin(lights[i].altitude))) * light_color.xyz * light_color.a;
color += compute_light(normal, lights[i].dir) * light_color.xyz * light_color.a;
}
color = filmic(color, 1. + f32(light_count));
+170 -101
View File
@@ -17,6 +17,20 @@
// Only read by `fs_color`'s shadow branch (custom-lights palette); the
// iteration pass (`fs_data`) never touches it.
@group(0) @binding(2) var<uniform> lights: array<Light, 16>;
// Only read by the adaptive-AA refine pass (`fs_refine`): the 1-sample-per-
// pixel data texture written by `fs_data`, which decides where to supersample.
@group(1) @binding(0) var coarse_tex: texture_2d<f32>;
// Pipeline-overridable specialization constants, set per pipeline from the
// uniforms' `kind` / `is_julia` / `de_coloring` (see `PipelineKey` in
// renderer.rs). Every per-iteration branch on them folds away at pipeline
// creation, so the hot loop only contains the current kind's math instead of
// testing all of them on every step. The matching uniform fields are still
// uploaded (the layout is shared with colorize.wgsl) but this shader must read
// these constants, never `u.kind` / `u.is_julia` / `u.de_coloring`.
override KIND: u32 = 0u;
override IS_JULIA: bool = false;
override DE: bool = false;
struct VsOut {
@builtin(position) pos: vec4<f32>,
@@ -56,50 +70,47 @@ fn diffabs(c: f32, d: f32) -> f32 {
return select(-d, 2.0 * c + d, cd > 0.0);
}
// Binomial coefficient C(n, k) as f32 (exact for the small powers we use).
fn binom(n: u32, k: u32) -> f32 {
var num = 1.0;
var den = 1.0;
for (var i: u32 = 0u; i < k; i = i + 1u) {
num = num * f32(n - i);
den = den * f32(i + 1u);
}
return num / den;
}
// Perturbation delta for z -> z^p: sum_{k=1}^{p} C(p,k) Z^{p-k} e^k. Expanded so
// the large z^p term is never formed (that would cancel catastrophically).
// Perturbation delta for z -> z^p: (Z+e)^p - Z^p = e * sum_{k=0}^{p-1} (Z+e)^k Z^{p-1-k}.
// The large z^p term is never formed (that would cancel catastrophically), and
// the sum is evaluated Horner-style (s <- s*(Z+e) + Z^j) so it needs neither a
// table of powers (a dynamically indexed local array spills to slow memory on
// most GPUs) nor binomial coefficients. Forming Z+e rounds e away when it's
// tiny, but that only perturbs `s` by a relative f32 epsilon, and the result
// is `e * s`, so the delta keeps full relative precision.
fn multibrot_delta(z: vec2<f32>, e: vec2<f32>, p: u32) -> vec2<f32> {
var zp: array<vec2<f32>, 9>; // Z^0 .. Z^8
zp[0] = vec2<f32>(1.0, 0.0);
for (var j: u32 = 1u; j <= p; j = j + 1u) {
zp[j] = cmul(zp[j - 1u], z);
let y = z + e;
var s = vec2<f32>(1.0, 0.0);
var zj = vec2<f32>(1.0, 0.0);
for (var j: u32 = 1u; j < p; j = j + 1u) {
zj = cmul(zj, z); // Z^j
s = cmul(s, y) + zj;
}
var acc = vec2<f32>(0.0, 0.0);
var ek = vec2<f32>(1.0, 0.0); // e^0
for (var k: u32 = 1u; k <= p; k = k + 1u) {
ek = cmul(ek, e); // e^k
acc = acc + binom(p, k) * cmul(zp[p - k], ek);
}
return acc;
return cmul(e, s);
}
// Number of terms kept in `complex_multibrot_delta`'s series. Truncation, not
// exactness: unlike `multibrot_delta` (a finite binomial sum for an integer
// power), a complex power has no finite expansion, so this converges rather
// than terminates. Fine as long as perturbation's usual invariant (|e| << |z|,
// Maximum number of terms in `complex_multibrot_delta`'s series (matches the
// `cm_coef` uniform array: 8 vec4s = 16 complex coefficients). Truncation, not
// exactness: unlike `multibrot_delta` (a finite sum for an integer power), a
// complex power has no finite expansion, so this converges rather than
// terminates. Fine as long as perturbation's usual invariant (|e| << |z|,
// kept true by rebasing) holds, since each extra term is O(w^k) smaller.
const COMPLEX_MULTIBROT_TERMS: u32 = 16u;
// Complex binomial coefficient C(p, k), k in 1..=16, precomputed on the CPU
// (they depend only on p; see `complex_binomials` in app.rs).
fn cm_coef(k: u32) -> vec2<f32> {
let v = u.cm_coef[(k - 1u) / 2u];
return select(v.xy, v.zw, (k & 1u) == 0u);
}
// Perturbation delta for z -> z^p with a complex p: (Z+e)^p - Z^p.
//
// When |e| << |Z| (the common case: it's the whole reason perturbation
// works), forming Z+e directly would round e away in f32, so instead expand
// = Z^p * ((1+w)^p - 1), w = e/Z, as a Taylor series in w: (1+w)^p - 1 =
// sum_{k=1}^N C(p,k) w^k, with the complex binomial coefficient built up
// incrementally: C(p,k) = C(p,k-1) * (p-(k-1)) / k. Unlike `multibrot_delta`
// (a finite binomial sum for an integer power), this only *converges* — and
// only for |w| < 1 — rather than terminating exactly.
// sum_{k=1}^N C(p,k) w^k. The series stops as soon as the next w^k is
// negligible against the running sum (below f32 precision) — at deep zoom w
// is tiny, so that's typically after 2-3 terms instead of all 16.
//
// Right after a rebase (or near a reference point close to zero, where w is
// singular), e is *not* small relative to Z — that's normal perturbation
@@ -112,13 +123,14 @@ fn complex_multibrot_delta(z: vec2<f32>, e: vec2<f32>, p: vec2<f32>) -> vec2<f32
let w2 = dot(e, e) / dot(z, z);
if w2 < 0.25 {
let w = cdiv(e, z);
var wk = vec2<f32>(1.0, 0.0); // w^0
var coef = vec2<f32>(1.0, 0.0); // C(p,0)
var wk = w; // w^1
var acc = vec2<f32>(0.0, 0.0);
for (var k: u32 = 1u; k <= COMPLEX_MULTIBROT_TERMS; k = k + 1u) {
coef = cdiv(cmul(coef, p - vec2<f32>(f32(k - 1u), 0.0)), vec2<f32>(f32(k), 0.0));
acc = acc + cmul(cm_coef(k), wk);
wk = cmul(wk, w);
acc = acc + cmul(coef, wk);
if dot(wk, wk) < 1e-18 * dot(acc, acc) {
break;
}
}
return cmul(cpow(z, p), acc);
}
@@ -129,7 +141,7 @@ fn complex_multibrot_delta(z: vec2<f32>, e: vec2<f32>, p: vec2<f32>) -> vec2<f32
// where `z` is the reference orbit value X_m. `step_add` (dc) is added by the
// caller. Must match `FractalKind` on the CPU side.
fn advance_delta(z: vec2<f32>, e: vec2<f32>) -> vec2<f32> {
if u.kind == KIND_BURNING_SHIP {
if KIND == KIND_BURNING_SHIP {
// (|x| + i|y|)^2 has real part x^2 - y^2 (an ordinary square delta) and
// imaginary part 2|x y|. The imaginary delta is 2(|x y| - |X Y|); diffabs
// computes it exactly, even where the product x y changes sign — which the
@@ -138,34 +150,34 @@ fn advance_delta(z: vec2<f32>, e: vec2<f32>) -> vec2<f32> {
let base = 2.0 * cmul(z, e) + cmul(e, e);
let dp = z.x * e.y + z.y * e.x + e.x * e.y;
return vec2<f32>(base.x, 2.0 * diffabs(z.x * z.y, dp));
} else if u.kind == KIND_TRICORN {
} else if KIND == KIND_TRICORN {
let cz = conj(z);
let ce = conj(e);
return 2.0 * cmul(cz, ce) + cmul(ce, ce);
} else if u.kind == KIND_MULTIBROT {
} else if KIND == KIND_MULTIBROT {
return multibrot_delta(z, e, clamp(u.power, 2u, 8u));
} else if u.kind == KIND_CELTIC {
} else if KIND == KIND_CELTIC {
// z^2 delta split: sq.x = delta of Re(z^2), sq.y = delta of Im(z^2).
// Celtic abs the real output, so |Re(z^2)| delta = diffabs(Re(Z^2), sq.x).
let sq = 2.0 * cmul(z, e) + cmul(e, e);
return vec2<f32>(diffabs(z.x * z.x - z.y * z.y, sq.x), sq.y);
} else if u.kind == KIND_BUFFALO {
} else if KIND == KIND_BUFFALO {
// Abs both outputs: real |Re(z^2)|, imag -|Im(z^2)| (Im(Z^2) = 2 X Y).
let sq = 2.0 * cmul(z, e) + cmul(e, e);
return vec2<f32>(diffabs(z.x * z.x - z.y * z.y, sq.x),
-diffabs(2.0 * z.x * z.y, sq.y));
} else if u.kind == KIND_PERPENDICULAR {
} else if KIND == KIND_PERPENDICULAR {
// real x^2 - y^2 (ordinary square delta), imag -2 x |y|.
// d(-2 x |y|) = -2[ X·(|Y+ey|-|Y|) + ex·|Y+ey| ]; diffabs gives |Y+ey|-|Y|.
let sq = 2.0 * cmul(z, e) + cmul(e, e);
let da = diffabs(z.y, e.y); // |Y + ey| - |Y|
let abs_yf = abs(z.y) + da; // |Y + ey|
return vec2<f32>(sq.x, -2.0 * (z.x * da + e.x * abs_yf));
} else if u.kind == KIND_LAMBDA {
} else if KIND == KIND_LAMBDA {
// Lambda map: z^{n+1} = λ·z·(1-z). Delta: e = λ·e·(1-2z-e).
let one_minus_2z_minus_e = vec2<f32>(1.0 - 2.0 * z.x - e.x, -2.0 * z.y - e.y);
return cmul(u.lambda_l, cmul(e, one_minus_2z_minus_e));
} else if u.kind == KIND_COMPLEX_MULTIBROT {
} else if KIND == KIND_COMPLEX_MULTIBROT {
return complex_multibrot_delta(z, e, u.complex_power);
}
return 2.0 * cmul(z, e) + cmul(e, e); // Mandelbrot (and Phoenix square part)
@@ -177,17 +189,17 @@ fn advance_delta(z: vec2<f32>, e: vec2<f32>) -> vec2<f32> {
// Burning Ship / Tricorn we use |f'| ~ |2Z|, which keeps the DE magnitude close
// enough to de-speckle filaments.
fn fprime(z: vec2<f32>) -> vec2<f32> {
if u.kind == KIND_MULTIBROT {
if KIND == KIND_MULTIBROT {
let p = clamp(u.power, 2u, 8u);
var zk = vec2<f32>(1.0, 0.0); // Z^0
for (var k: u32 = 1u; k < p; k = k + 1u) {
var zk = z; // Z^1
for (var k: u32 = 2u; k < p; k = k + 1u) {
zk = cmul(zk, z); // -> Z^{p-1}
}
return f32(p) * zk;
} else if u.kind == KIND_LAMBDA {
} else if KIND == KIND_LAMBDA {
// Lambda: f'(z) = λ·(1-2z).
return cmul(u.lambda_l, vec2<f32>(1.0 - 2.0 * z.x, -2.0 * z.y));
} else if u.kind == KIND_COMPLEX_MULTIBROT {
} else if KIND == KIND_COMPLEX_MULTIBROT {
// f'(z) = p * z^(p-1).
return cmul(u.complex_power, cpow(z, u.complex_power - vec2<f32>(1.0, 0.0)));
}
@@ -209,6 +221,10 @@ struct Sample {
// starts at 0); for Julia it is the z-plane offset that seeds the initial delta
// (c is fixed, so nothing is added per step).
fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
// Loop invariants, read once instead of on every iteration.
let max_iter = u.max_iter;
let bailout_sq = u.bailout_sq;
let ref_len = u.ref_len;
let z0 = ref_orbit[0]; // reference start (0 for Mandelbrot, center for Julia)
// Main cardioid / period-2 bulb bypass: those points never escape, so skip
@@ -217,7 +233,7 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
// orbit itself, since X_1 = X_0^2 + C_ref = C_ref. That's only f32-accurate,
// so skip the test once a pixel is smaller than that error (deep zoom),
// where it could misclassify pixels right at the boundary.
if u.kind == KIND_MANDELBROT && u.is_julia == 0u && u.ref_len > 1u && px > 1e-6 {
if KIND == KIND_MANDELBROT && !IS_JULIA && ref_len > 1u && px > 1e-6 {
let c = ref_orbit[1] + offset;
let xq = c.x - 0.25;
let q = xq * xq + c.y * c.y;
@@ -229,47 +245,48 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
}
}
// Set plane: delta starts at 0 and gains dc every step. Julia: the offset
// seeds the delta and nothing is added per step.
var step_add = offset;
var e = vec2<f32>(0.0, 0.0);
// Orbit derivative for distance estimation. For the set plane it is d/dc
// (starts at 0, gains +1 each step); for Julia it is d/dz0 (starts at 1).
var dz = vec2<f32>(0.0, 0.0);
var dz_seed = vec2<f32>(1.0, 0.0);
if IS_JULIA {
step_add = vec2<f32>(0.0, 0.0);
e = offset;
dz = vec2<f32>(1.0, 0.0);
}
// Previous-iterate state for the Phoenix two-term recurrence (delta of
// y_{n-1}, and its derivative for DE). Both start at 0 (y_{-1} = 0).
var e_prev = vec2<f32>(0.0, 0.0);
var dz_prev = vec2<f32>(0.0, 0.0);
if u.is_julia != 0u {
step_add = vec2<f32>(0.0, 0.0);
e = offset;
dz = vec2<f32>(1.0, 0.0);
dz_seed = vec2<f32>(0.0, 0.0);
}
var m: u32 = 0u; // reference index; invariant: y_n = X[m] + e
var m: u32 = 0u; // reference index; invariant: y_n = xm + e, xm = X[m]
var n: u32 = 0u; // total iteration count
var z = vec2<f32>(0.0, 0.0); // full value y_n, kept for coloring
var xm = z0; // X[m], carried so each step loads the orbit once
var z = xm + e; // full value y_n, kept for coloring
var z2 = dot(z, z);
var escaped = false;
loop {
let xm = ref_orbit[m];
z = xm + e;
let z2 = dot(z, z);
if z2 > u.bailout_sq {
if z2 > bailout_sq {
escaped = true;
break;
}
if n >= u.max_iter {
if n >= max_iter {
break; // interior
}
// Propagate the derivative of the full orbit (unaffected by rebasing,
// which only re-expresses the same value). Only when DE is enabled.
// Phoenix's two-term map adds p·dz_{n-1} and carries the previous dz.
if u.de_coloring != 0u {
var dz_new = cmul(fprime(z), dz) + dz_seed;
if u.kind == KIND_PHOENIX {
if DE {
var dz_new = cmul(fprime(z), dz);
if !IS_JULIA {
dz_new.x = dz_new.x + 1.0;
}
if KIND == KIND_PHOENIX {
dz_new = dz_new + cmul(u.phoenix_p, dz_prev);
dz_prev = dz;
}
@@ -279,8 +296,9 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
// Advance the delta by this fractal's formula (+ dc for the set plane).
// Phoenix additionally adds p·e_{n-1} and carries the previous delta.
let e_old = e;
let z_old = z;
e = advance_delta(xm, e) + step_add;
if u.kind == KIND_PHOENIX {
if KIND == KIND_PHOENIX {
e = e + cmul(u.phoenix_p, e_prev);
e_prev = e_old;
}
@@ -288,23 +306,27 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
n = n + 1u;
// Keep the reference index valid and the delta small.
if m >= u.ref_len {
if m >= ref_len {
// Reference exhausted: any pixel that followed it this far has
// effectively escaped (interior pixels rebase before reaching here).
z = ref_orbit[u.ref_len - 1u] + e;
z = xm + e;
escaped = true;
break;
}
let y = ref_orbit[m] + e;
if dot(y, y) < dot(e, e) {
xm = ref_orbit[m];
z = xm + e;
z2 = dot(z, z);
if z2 < dot(e, e) {
// Rebase to index 0: carry the full value as the new delta. Valid
// because y_n = X[0] + (y_n - X[0]); for Mandelbrot X[0]=0.
// because y_n = X[0] + (y_n - X[0]); for Mandelbrot X[0]=0. The
// full value `z` (and `z2`) is unchanged by the re-expression.
// Phoenix: after rebasing the implied previous reference is Y[-1]=0,
// so the previous delta becomes the full previous value y_n (= z).
if u.kind == KIND_PHOENIX {
e_prev = z;
// so the previous delta becomes the full previous value y_{n-1}.
if KIND == KIND_PHOENIX {
e_prev = z_old;
}
e = y - z0;
e = z - z0;
xm = z0;
m = 0u;
}
}
@@ -313,11 +335,11 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
return Sample(0.0, 1.0, false); // interior of the set
}
let z2 = dot(z, z);
z2 = dot(z, z);
// Continuous (smooth) iteration count.
let log_zn = 0.5 * log(max(z2, 1.0));
let nu = log2(log_zn / log(2.0));
let nu = log2(log_zn * INV_LN2);
let smooth_i = f32(n) + 1.0 - nu;
// sqrt compresses the huge iteration counts of deep zooms so the palette
@@ -325,7 +347,7 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
let ci = sqrt(max(smooth_i, 0.0));
var de = 1.0;
if u.de_coloring != 0u {
if DE {
// Exterior distance estimate (complex-plane units): |z|·ln|z| / |dz|.
// Divided by the pixel footprint it becomes a distance in pixels; we
// darken toward the boundary (< ~1 px away) so filaments stay crisp
@@ -334,15 +356,15 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
let zmag = sqrt(max(z2, 1.0));
let dzmag = sqrt(max(dot(dz, dz), 1e-20));
let d = zmag * log(zmag) / dzmag;
var max_de = 1.;
if u.shadow != 0u {
max_de = 1000.;
}
let max_de = select(1.0, 1000.0, u.shadow != 0u);
de = clamp(d / max(px, 1e-30), 0.0, max_de);
}
return Sample(ci, de, true);
}
// 1 / ln(2), for the smooth iteration count's log2(ln|z| / ln 2).
const INV_LN2: f32 = 1.4426950408889634;
// Map a sample's escape data through the palette (+ DE darkening). This is the
// only color-dependent step, so it can be redone without re-iterating. Interior
// samples are black.
@@ -353,13 +375,13 @@ fn color_sample(s: Sample) -> vec3<f32> {
return classic_color(s.ci, s.de);
}
// Supersampled escape data at one point: average (ci, DE factor) over the
// AA grid's escaped sub-samples, plus the fraction that landed in the
// interior. Shared by `fs_data` (writes it straight to the data texture) and
// `fs_color`'s shadow branch (used both at the pixel and at its two
// neighbours, to build a DE height field without a texture round-trip).
fn aggregate_sample(base: vec2<f32>, dx: vec2<f32>, dy: vec2<f32>, px: f32) -> vec3<f32> {
let aa = max(u.aa_level, 1u);
// Supersampled escape data at one point: average (ci, DE factor) over an
// `aa`×`aa` grid's escaped sub-samples, plus the fraction that landed in the
// interior. Shared by `fs_data` (1 sample), `fs_refine` (the AA grid, only on
// pixels that need it) and `fs_color`'s shadow branch (used both at the pixel
// and at its two neighbours, to build a DE height field without a texture
// round-trip).
fn aggregate_sample(base: vec2<f32>, dx: vec2<f32>, dy: vec2<f32>, px: f32, aa: u32) -> vec3<f32> {
let inv = 1.0 / f32(aa);
var ci_sum = 0.0;
var de_sum = 0.0;
@@ -386,8 +408,8 @@ fn aggregate_sample(base: vec2<f32>, dx: vec2<f32>, dy: vec2<f32>, px: f32) -> v
// Iteration pass: write per-pixel escape data (color-independent) so a colour
// change is remapped by the cheap colourise pass without re-iterating.
// R = ci (palette parameter), G = DE factor, B = interior fraction (for AA).
// AA is grid-supersampled here; the interior fraction lets the colourise pass
// anti-alias the set boundary (blend toward black) after the fact.
// Always one sample per pixel: anti-aliasing is added afterwards, only where
// it matters, by `fs_refine`.
@fragment
fn fs_data(in: VsOut) -> @location(0) vec4<f32> {
let base = in.centered * u.span + u.dc_offset;
@@ -395,35 +417,82 @@ fn fs_data(in: VsOut) -> @location(0) vec4<f32> {
let dy = dpdy(base);
let px = length(abs(dx) + abs(dy));
return vec4<f32>(aggregate_sample(base, dx, dy, px), 1.0);
return vec4<f32>(aggregate_sample(base, dx, dy, px, 1u), 1.0);
}
// Adaptive-AA thresholds for `fs_refine`: a pixel is supersampled only if a
// 4-neighbour's 1-spp sample differs from its own by more than this. `ci`
// steps are palette-phase steps of `ci * color_scale` (color_scale <= 1 in the
// UI), so 0.02 keeps anything visibly banded; DE is compared relative to its
// own magnitude (it's in pixels, up to 1000 for shadow/3D height fields).
const AA_CI_EPS: f32 = 0.02;
const AA_DE_EPS: f32 = 0.1;
fn aa_differs(c: vec4<f32>, n: vec4<f32>) -> bool {
if c.b != n.b {
return true; // interior / exterior boundary
}
if c.b != 0.0 {
return false; // both interior: uniformly black
}
return abs(n.r - c.r) > AA_CI_EPS || abs(n.g - c.g) > AA_DE_EPS * max(c.g, 0.1);
}
// Adaptive anti-aliasing pass (only run when AA is on): reads `fs_data`'s
// 1-spp texture and re-iterates the full AA grid only for pixels whose
// neighbourhood isn't smooth (set boundary, filaments, palette discontinuities).
// Everywhere else the centre sample already equals the grid average to within
// the thresholds above, so it's copied — which skips the AA cost entirely for
// the interior (the most expensive pixels, each burning max_iter) and for the
// smooth exterior.
@fragment
fn fs_refine(in: VsOut) -> @location(0) vec4<f32> {
// Derivatives first, while control flow is still uniform.
let base = in.centered * u.span + u.dc_offset;
let dx = dpdx(base);
let dy = dpdy(base);
let px = length(abs(dx) + abs(dy));
let p = vec2<i32>(in.pos.xy);
let hi = vec2<i32>(textureDimensions(coarse_tex)) - vec2<i32>(1, 1);
let c = textureLoad(coarse_tex, p, 0);
let l = textureLoad(coarse_tex, max(p - vec2<i32>(1, 0), vec2<i32>(0, 0)), 0);
let r = textureLoad(coarse_tex, min(p + vec2<i32>(1, 0), hi), 0);
let t = textureLoad(coarse_tex, max(p - vec2<i32>(0, 1), vec2<i32>(0, 0)), 0);
let b = textureLoad(coarse_tex, min(p + vec2<i32>(0, 1), hi), 0);
if aa_differs(c, l) || aa_differs(c, r) || aa_differs(c, t) || aa_differs(c, b) {
return vec4<f32>(aggregate_sample(base, dx, dy, px, max(u.aa_level, 1u)), 1.0);
}
return c;
}
// Combined iterate + colour in a single pass, for PNG export (which never needs
// incremental recolouring). The interactive path uses fs_data + the colourise
// pass so colour changes skip iteration.
// incremental recolouring). The interactive path uses fs_data (+ fs_refine) +
// the colourise pass so colour changes skip iteration. Export always runs the
// full AA grid on every pixel, for maximum quality.
@fragment
fn fs_color(in: VsOut) -> @location(0) vec4<f32> {
let base = in.centered * u.span + u.dc_offset;
let dx = dpdx(base);
let dy = dpdy(base);
let px = length(abs(dx) + abs(dy));
let aa = max(u.aa_level, 1u);
if u.shadow != 0u {
// No data texture to sample neighbours from (this pass never runs
// one), so build the same DE height field colorize.wgsl reads from
// the texture by aggregating live, at the pixel and its two
// neighbours a `dx`/`dy` step away.
let here = aggregate_sample(base, dx, dy, px);
let here = aggregate_sample(base, dx, dy, px, aa);
if here.z != 0.0 {
return vec4<f32>(0.1, 0.1, 0.1, 1.0);
}
let right = aggregate_sample(base + dx, dx, dy, px);
let down = aggregate_sample(base + dy, dx, dy, px);
let right = aggregate_sample(base + dx, dx, dy, px, aa);
let down = aggregate_sample(base + dy, dx, dy, px, aa);
let normal = normal_from_heights(here.y, right.y, down.y);
return vec4<f32>(shadow_color(normal), 1.0);
}
let aa = max(u.aa_level, 1u);
let inv = 1.0 / f32(aa);
var acc = vec3<f32>(0.0, 0.0, 0.0);
for (var sy: u32 = 0u; sy < aa; sy = sy + 1u) {
+101 -15
View File
@@ -3,7 +3,7 @@
//! shader with the same `naga` version wgpu uses — catching shader errors
//! without needing a GPU or a display.
fn validate(name: &str, src: &str) {
fn validate(name: &str, src: &str) -> (naga::Module, naga::valid::ModuleInfo) {
let module = match naga::front::wgsl::parse_str(src) {
Ok(m) => m,
Err(e) => panic!("{name}: WGSL parse error:\n{}", e.emit_to_string(src)),
@@ -12,21 +12,92 @@ fn validate(name: &str, src: &str) {
naga::valid::ValidationFlags::all(),
naga::valid::Capabilities::all(),
);
if let Err(e) = validator.validate(&module) {
panic!("{name}: WGSL validation error:\n{}", e.emit_to_string(src));
match validator.validate(&module) {
Ok(info) => (module, info),
Err(e) => panic!("{name}: WGSL validation error:\n{}", e.emit_to_string(src)),
}
}
#[test]
fn mandelbrot_shader_is_valid() {
validate(
"mandelbrot.wgsl",
concat!(
/// Number of fractal kinds, i.e. the `const KIND_*` declarations in
/// common.wgsl (one per `FractalKind` variant, values 0..N).
fn kind_count() -> u32 {
let n = include_str!("../src/shaders/common.wgsl")
.lines()
.filter(|l| l.starts_with("const KIND_"))
.count() as u32;
assert!(n >= 10, "found only {n} KIND_* constants in common.wgsl");
n
}
/// Specialize `module`'s `override`s with `constants` for `entry_point` (as
/// wgpu does at pipeline creation) and compile the result to SPIR-V, so a
/// shader that only breaks once a particular override value folds a branch
/// in or out is still caught.
fn specialize(
name: &str,
module: &naga::Module,
info: &naga::valid::ModuleInfo,
stage: naga::ShaderStage,
entry_point: &str,
constants: &[(&str, f64)],
) {
let mut pc = naga::back::PipelineConstants::default();
for (k, v) in constants {
pc.insert((*k).to_string(), *v);
}
let (module, info) = naga::back::pipeline_constants::process_overrides(
module,
info,
Some((stage, entry_point)),
&pc,
)
.unwrap_or_else(|e| panic!("{name} {entry_point} {constants:?}: override error: {e:?}"));
let pipeline = naga::back::spv::PipelineOptions {
shader_stage: stage,
entry_point: entry_point.to_string(),
};
naga::back::spv::write_vec(
&module,
&info,
&naga::back::spv::Options::default(),
Some(&pipeline),
)
.unwrap_or_else(|e| panic!("{name} {entry_point} {constants:?}: SPIR-V error: {e:?}"));
}
const MANDELBROT_SRC: &str = concat!(
include_str!("../src/shaders/common.wgsl"),
include_str!("../src/shaders/iterate_uniforms.wgsl"),
include_str!("../src/shaders/mandelbrot.wgsl"),
),
);
#[test]
fn mandelbrot_shader_is_valid() {
validate("mandelbrot.wgsl", MANDELBROT_SRC);
}
/// Every specialization renderer.rs can build (`PipelineKey`: kind × Julia ×
/// DE), for every fragment entry point.
#[test]
fn mandelbrot_shader_specializations_compile() {
let (module, info) = validate("mandelbrot.wgsl", MANDELBROT_SRC);
for kind in 0..kind_count() {
for julia in [0.0, 1.0] {
for de in [0.0, 1.0] {
let constants = [("KIND", kind as f64), ("IS_JULIA", julia), ("DE", de)];
for entry in ["fs_data", "fs_refine", "fs_color"] {
specialize(
"mandelbrot.wgsl",
&module,
&info,
naga::ShaderStage::Fragment,
entry,
&constants,
);
}
}
}
}
}
#[test]
@@ -52,13 +123,28 @@ fn blit_shader_is_valid() {
);
}
#[test]
fn buddhabrot_shader_is_valid() {
validate(
"buddhabrot.wgsl",
concat!(
const BUDDHABROT_SRC: &str = concat!(
include_str!("../src/shaders/common.wgsl"),
include_str!("../src/shaders/buddhabrot.wgsl"),
),
);
#[test]
fn buddhabrot_shader_is_valid() {
validate("buddhabrot.wgsl", BUDDHABROT_SRC);
}
/// Every per-kind accumulation pipeline buddhabrot.rs can build.
#[test]
fn buddhabrot_shader_specializations_compile() {
let (module, info) = validate("buddhabrot.wgsl", BUDDHABROT_SRC);
for kind in 0..kind_count() {
specialize(
"buddhabrot.wgsl",
&module,
&info,
naga::ShaderStage::Compute,
"cs_main",
&[("KIND", kind as f64)],
);
}
}