perf: improve overall performance

This commit is contained in:
2026-09-24 18:23:19 +02:00
parent 7dd99cc1af
commit a3fd152dff
13 changed files with 1070 additions and 412 deletions
+16 -10
View File
@@ -58,6 +58,12 @@ struct Uniforms {
complex_power: vec2<f32>,
};
// Fractal kind, as a pipeline-overridable constant (set per compute pipeline
// from `u.kind`, see `BuddhabrotRenderer::compute_pipeline`): every kind
// branch in the iteration loop folds away at pipeline creation. Read this,
// never `u.kind`.
override KIND: u32 = 0u;
const PALETTE_NEBULA: u32 = 0u;
const PALETTE_YELLOW: u32 = 1u;
const PALETTE_GRAYSCALE: u32 = 2u;
@@ -95,25 +101,25 @@ fn complex_pow(z: vec2<f32>, p: u32) -> vec2<f32> {
// Must match `FractalKind` in reference.rs (the direct, non-perturbative form
// of the same formulas).
fn advance(z: vec2<f32>, zp: vec2<f32>, c: vec2<f32>) -> vec2<f32> {
if u.kind == KIND_BURNING_SHIP {
if KIND == KIND_BURNING_SHIP {
return vec2<f32>(z.x * z.x - z.y * z.y, 2.0 * abs(z.x * z.y)) + c;
} else if u.kind == KIND_TRICORN {
} else if KIND == KIND_TRICORN {
return vec2<f32>(z.x * z.x - z.y * z.y, -2.0 * z.x * z.y) + c;
} else if u.kind == KIND_MULTIBROT {
} else if KIND == KIND_MULTIBROT {
return complex_pow(z, clamp(u.power, 2u, 8u)) + c;
} else if u.kind == KIND_CELTIC {
} else if KIND == KIND_CELTIC {
return vec2<f32>(abs(z.x * z.x - z.y * z.y), 2.0 * z.x * z.y) + c;
} else if u.kind == KIND_PERPENDICULAR {
} else if KIND == KIND_PERPENDICULAR {
return vec2<f32>(z.x * z.x - z.y * z.y, -2.0 * z.x * abs(z.y)) + c;
} else if u.kind == KIND_BUFFALO {
} else if KIND == KIND_BUFFALO {
return vec2<f32>(abs(z.x * z.x - z.y * z.y), -abs(2.0 * z.x * z.y)) + c;
} else if u.kind == KIND_PHOENIX {
} else if KIND == KIND_PHOENIX {
let sq = vec2<f32>(z.x * z.x - z.y * z.y, 2.0 * z.x * z.y);
return sq + c + cmul(u.phoenix_p, zp);
} else if u.kind == KIND_LAMBDA {
} else if KIND == KIND_LAMBDA {
// l * z * (1 - z); c is unused (see file doc comment above).
return cmul(u.lambda_l, cmul(z, vec2<f32>(1.0 - z.x, -z.y)));
} else if u.kind == KIND_COMPLEX_MULTIBROT {
} else if KIND == KIND_COMPLEX_MULTIBROT {
return cpow(z, u.complex_power) + c;
}
return vec2<f32>(z.x * z.x - z.y * z.y, 2.0 * z.x * z.y) + c; // Mandelbrot
@@ -176,7 +182,7 @@ fn cs_main(@builtin(global_invocation_id) gid: vec3<u32>) {
var c = sample;
var z0 = vec2<f32>(0.0, 0.0);
if u.kind == KIND_LAMBDA {
if KIND == KIND_LAMBDA {
c = vec2<f32>(0.0, 0.0); // unused by the Lambda step
z0 = sample;
}
+49 -41
View File
@@ -20,34 +20,34 @@ fn vs_main(@builtin(vertex_index) idx: u32) -> @builtin(position) vec4<f32> {
}
fn shadow_fragment(pos: vec2<f32>) -> vec4<f32> {
let x = i32(pos.x);
let y = i32(pos.y);
let size = textureDimensions(data_tex);
if textureLoad(data_tex, vec2<i32>(x, y), 0).b != 0. {
let here = textureLoad(data_tex, vec2<i32>(x, y), 0);
if here.b != 0. {
return vec4<f32>(0.1, 0.1, 0.1, 1.0);
} else {
// Forward differences, except on the last column/row where x+1 / y+1
// is off the texture: fall back to a backward difference, mirrored
// (h0 + (h0 - h[-1])) so the slope keeps the sign normal_from_heights
// expects — plugging h[-1] in directly would flip the normal there.
let h0 = textureLoad(data_tex, vec2<i32>(x, y), 0).g;
var h1: f32;
if x + 1 < i32(size.x) {
h1 = textureLoad(data_tex, vec2<i32>(x + 1, y), 0).g;
} else {
h1 = 2.0 * h0 - textureLoad(data_tex, vec2<i32>(x - 1, y), 0).g;
}
var h2: f32;
if y + 1 < i32(size.y) {
h2 = textureLoad(data_tex, vec2<i32>(x, y + 1), 0).g;
} else {
h2 = 2.0 * h0 - textureLoad(data_tex, vec2<i32>(x, y - 1), 0).g;
}
let normal = normal_from_heights(h0, h1, h2);
return vec4<f32>(shadow_color(normal), 1.0);
}
// Forward differences, except on the last column/row where x+1 / y+1
// is off the texture: fall back to a backward difference, mirrored
// (h0 + (h0 - h[-1])) so the slope keeps the sign normal_from_heights
// expects — plugging h[-1] in directly would flip the normal there.
let h0 = here.g;
var h1: f32;
if x + 1 < i32(size.x) {
h1 = textureLoad(data_tex, vec2<i32>(x + 1, y), 0).g;
} else {
h1 = 2.0 * h0 - textureLoad(data_tex, vec2<i32>(x - 1, y), 0).g;
}
var h2: f32;
if y + 1 < i32(size.y) {
h2 = textureLoad(data_tex, vec2<i32>(x, y + 1), 0).g;
} else {
h2 = 2.0 * h0 - textureLoad(data_tex, vec2<i32>(x, y - 1), 0).g;
}
let normal = normal_from_heights(h0, h1, h2);
return vec4<f32>(shadow_color(normal), 1.0);
}
@fragment
fn fs_main(@builtin(position) pos: vec4<f32>) -> @location(0) vec4<f32> {
if u.shadow == 2u {
@@ -68,19 +68,25 @@ fn fs_main(@builtin(position) pos: vec4<f32>) -> @location(0) vec4<f32> {
}
}
fn sdf(pos: vec3<f32>) -> f32 {
let aspect_ratio = u.screen_dim.x / u.screen_dim.y;
let size = vec2<f32>(textureDimensions(data_tex));
var texture_pos_f32 = vec2<f32>(pos.x * size.x / aspect_ratio, pos.y * size.y);
var texture_pos = vec2<i32>(i32(texture_pos_f32.x), i32(texture_pos_f32.y));
texture_pos.x = clamp(texture_pos.x, 0, i32(size.x) - 1);
texture_pos.y = clamp(texture_pos.y, 0, i32(size.y) - 1);
// Per-frame constants of the raymarch, computed once per pixel in
// `ray_marching` rather than on each of the up-to-100 `sdf` steps.
struct MarchConsts {
size: vec2<f32>,
// (size.x / aspect_ratio, size.y): world xy -> texel scale.
to_texel: vec2<f32>,
size_i: vec2<i32>,
inv_size_y: f32,
};
let to_texture = max(-min(texture_pos_f32, vec2(0.)), max(texture_pos_f32 - size, vec2(0.)));
let dist_to_texture = length(to_texture) / size.y;
fn sdf(pos: vec3<f32>, k: MarchConsts) -> f32 {
let texture_pos_f32 = pos.xy * k.to_texel;
let texture_pos = clamp(vec2<i32>(texture_pos_f32), vec2<i32>(0, 0), k.size_i - vec2<i32>(1, 1));
let to_texture = max(-min(texture_pos_f32, vec2(0.)), max(texture_pos_f32 - k.size, vec2(0.)));
let dist_to_texture = length(to_texture) * k.inv_size_y;
let px = textureLoad(data_tex, texture_pos, 0);
let de = (px.g / size.y) * 0.5;
let de = (px.g * k.inv_size_y) * 0.5;
// Height is measured toward -z, the side the camera sits on (it looks
// along +z), so the terrain is solid on +z: interior plateau at z = 0,
// exterior sloping away from the camera as `de` grows.
@@ -105,7 +111,11 @@ fn sdf(pos: vec3<f32>) -> f32 {
}
fn ray_marching(pos: vec4<f32>) -> vec4<f32> {
let size = vec2<f32>(textureDimensions(data_tex));
let size_i = vec2<i32>(textureDimensions(data_tex));
let size = vec2<f32>(size_i);
let aspect_ratio = u.screen_dim.x / u.screen_dim.y;
let k = MarchConsts(size, vec2<f32>(size.x / aspect_ratio, size.y), size_i, 1.0 / size.y);
let in_texture = vec2<f32>(
(pos.x / size.x) * 2. - 1.,
(pos.y / size.y) * 2. - 1.,
@@ -123,22 +133,20 @@ fn ray_marching(pos: vec4<f32>) -> vec4<f32> {
let dist_threshold = 0.000001;
while i < 100u {
if length(p - ray_origin) > 3. {
let from_origin = p - ray_origin;
if dot(from_origin, from_origin) > 9. {
break;
}
dist = sdf(p);
dist = sdf(p, k);
if dist < dist_threshold {
break;
}
p += dist * ray_dir;
i += 1u;
p += dist * ray_dir;
i += 1u;
}
if dist < dist_threshold {
let aspect_ratio = u.screen_dim.x / u.screen_dim.y;
let size = vec2<f32>(textureDimensions(data_tex));
let texture_pos_f32 = vec2<f32>(p.x * size.x / aspect_ratio, p.y * size.y);
return shadow_fragment(texture_pos_f32);
return shadow_fragment(p.xy * k.to_texel);
}
return vec4<f32>(1., 0., 0., 1.);
}
+23 -22
View File
@@ -35,11 +35,18 @@ struct Uniforms {
shadow: u32,
// camera direction vector
camera_direction: vec3<f32>,
// Number of live entries at the start of `lights` (fills the vec3's tail
// padding slot).
light_count: u32,
// inverse of the camera's view-projection matrix, for reconstructing a
// world-space ray origin per pixel in the raymarcher
camera_inv_proj: mat4x4<f32>,
// Screen dimensions
screen_dim: vec2<f32>
screen_dim: vec2<f32>,
// Complex binomial coefficients C(complex_power, k) for k = 1..16, two per
// vec4 (k odd in .xy, k even in .zw), for the Complex Multibrot delta
// series. Precomputed on the CPU since they only depend on the power.
cm_coef: array<vec4<f32>, 8>,
};
// Smooth cyclic palettes (Inigo Quilez cosine palettes), selected by id.
@@ -74,20 +81,21 @@ fn classic_color(ci: f32, de: f32) -> vec3<f32> {
return palette(u.palette_id, t) * sqrt(de);
}
// A single directional/point light, set by the UI's light list. `color`'s
// alpha channel doubles as intensity (see `shadow_color`'s use of
// `light_color.a`). Each shader that binds a `lights: array<Light, 16>`
// uniform (colorize.wgsl, mandelbrot.wgsl's export shadow path) uses this
// same layout.
// A single directional light, built on the CPU from the UI's light list
// (`GpuLight` in lights.rs): `dir` is the unit direction toward the light
// (precomputed from azimuth/altitude so the shader does no trig), `color` a
// packed RGBA8 whose alpha doubles as intensity. Only the first
// `u.light_count` entries are live, all with a non-zero colour. Each shader
// that binds a `lights: array<Light, 16>` uniform (colorize.wgsl,
// mandelbrot.wgsl's export shadow path) uses this same layout.
struct Light {
azimuth: f32,
altitude: f32,
dir: vec3<f32>,
color: u32,
_pad: u32,
};
// Lambertian term for a unit `light` direction.
fn compute_light(normal: vec3<f32>, light: vec3<f32>) -> vec3<f32> {
return vec3<f32>(max(0., dot(normal, normalize(light))));
return vec3<f32>(max(0., dot(normal, light)));
}
fn uncharted2tonemap(x: vec3<f32>) -> vec3<f32> {
@@ -139,27 +147,20 @@ fn normal_from_heights(h0: f32, h1: f32, h2: f32) -> vec3<f32> {
fn shadow_color(normal: vec3<f32>) -> vec3<f32> {
var color: vec3<f32>;
if u.shadow_palette_id == 0u {
color = compute_light(normal, vec3<f32>(.5, .5, .5)) + vec3<f32>(0.58, 0.85, 1.) * 0.2;
color = compute_light(normal, vec3<f32>(0.57735027, 0.57735027, 0.57735027)) + vec3<f32>(0.58, 0.85, 1.) * 0.2;
color = filmic(color, 2.5);
color = contrast(color, 4., 0.67);
} else if u.shadow_palette_id == 1u {
color = compute_light(normal, vec3<f32>(0., .5, .5)) * vec3<f32>(1., 0.5, 0.5) + compute_light(normal, vec3<f32>(0.5, 0., .5)) * vec3<f32>(0.5, 1., 1.);
color = compute_light(normal, vec3<f32>(0., 0.70710678, 0.70710678)) * vec3<f32>(1., 0.5, 0.5) + compute_light(normal, vec3<f32>(0.70710678, 0., 0.70710678)) * vec3<f32>(0.5, 1., 1.);
color = filmic(color, 4.2);
} else {
color = vec3<f32>(0);
var light_count = 0;
for (var i = 0u; i < 16; i++) {
let light_count = min(u.light_count, 16u);
for (var i = 0u; i < light_count; i++) {
let light_color = unpack4x8unorm(lights[i].color);
if any(light_color != vec4<f32>(0)) {
light_count += 1;
}
color += compute_light(normal, vec3<f32>(
cos(lights[i].azimuth) * cos(lights[i].altitude),
sin(lights[i].azimuth) * cos(lights[i].altitude),
sin(lights[i].altitude))) * light_color.xyz * light_color.a;
color += compute_light(normal, lights[i].dir) * light_color.xyz * light_color.a;
}
color = filmic(color, 1. + f32(light_count));
+195 -126
View File
@@ -17,6 +17,20 @@
// Only read by `fs_color`'s shadow branch (custom-lights palette); the
// iteration pass (`fs_data`) never touches it.
@group(0) @binding(2) var<uniform> lights: array<Light, 16>;
// Only read by the adaptive-AA refine pass (`fs_refine`): the 1-sample-per-
// pixel data texture written by `fs_data`, which decides where to supersample.
@group(1) @binding(0) var coarse_tex: texture_2d<f32>;
// Pipeline-overridable specialization constants, set per pipeline from the
// uniforms' `kind` / `is_julia` / `de_coloring` (see `PipelineKey` in
// renderer.rs). Every per-iteration branch on them folds away at pipeline
// creation, so the hot loop only contains the current kind's math instead of
// testing all of them on every step. The matching uniform fields are still
// uploaded (the layout is shared with colorize.wgsl) but this shader must read
// these constants, never `u.kind` / `u.is_julia` / `u.de_coloring`.
override KIND: u32 = 0u;
override IS_JULIA: bool = false;
override DE: bool = false;
struct VsOut {
@builtin(position) pos: vec4<f32>,
@@ -56,50 +70,47 @@ fn diffabs(c: f32, d: f32) -> f32 {
return select(-d, 2.0 * c + d, cd > 0.0);
}
// Binomial coefficient C(n, k) as f32 (exact for the small powers we use).
fn binom(n: u32, k: u32) -> f32 {
var num = 1.0;
var den = 1.0;
for (var i: u32 = 0u; i < k; i = i + 1u) {
num = num * f32(n - i);
den = den * f32(i + 1u);
}
return num / den;
}
// Perturbation delta for z -> z^p: sum_{k=1}^{p} C(p,k) Z^{p-k} e^k. Expanded so
// the large z^p term is never formed (that would cancel catastrophically).
// Perturbation delta for z -> z^p: (Z+e)^p - Z^p = e * sum_{k=0}^{p-1} (Z+e)^k Z^{p-1-k}.
// The large z^p term is never formed (that would cancel catastrophically), and
// the sum is evaluated Horner-style (s <- s*(Z+e) + Z^j) so it needs neither a
// table of powers (a dynamically indexed local array spills to slow memory on
// most GPUs) nor binomial coefficients. Forming Z+e rounds e away when it's
// tiny, but that only perturbs `s` by a relative f32 epsilon, and the result
// is `e * s`, so the delta keeps full relative precision.
fn multibrot_delta(z: vec2<f32>, e: vec2<f32>, p: u32) -> vec2<f32> {
var zp: array<vec2<f32>, 9>; // Z^0 .. Z^8
zp[0] = vec2<f32>(1.0, 0.0);
for (var j: u32 = 1u; j <= p; j = j + 1u) {
zp[j] = cmul(zp[j - 1u], z);
let y = z + e;
var s = vec2<f32>(1.0, 0.0);
var zj = vec2<f32>(1.0, 0.0);
for (var j: u32 = 1u; j < p; j = j + 1u) {
zj = cmul(zj, z); // Z^j
s = cmul(s, y) + zj;
}
var acc = vec2<f32>(0.0, 0.0);
var ek = vec2<f32>(1.0, 0.0); // e^0
for (var k: u32 = 1u; k <= p; k = k + 1u) {
ek = cmul(ek, e); // e^k
acc = acc + binom(p, k) * cmul(zp[p - k], ek);
}
return acc;
return cmul(e, s);
}
// Number of terms kept in `complex_multibrot_delta`'s series. Truncation, not
// exactness: unlike `multibrot_delta` (a finite binomial sum for an integer
// power), a complex power has no finite expansion, so this converges rather
// than terminates. Fine as long as perturbation's usual invariant (|e| << |z|,
// Maximum number of terms in `complex_multibrot_delta`'s series (matches the
// `cm_coef` uniform array: 8 vec4s = 16 complex coefficients). Truncation, not
// exactness: unlike `multibrot_delta` (a finite sum for an integer power), a
// complex power has no finite expansion, so this converges rather than
// terminates. Fine as long as perturbation's usual invariant (|e| << |z|,
// kept true by rebasing) holds, since each extra term is O(w^k) smaller.
const COMPLEX_MULTIBROT_TERMS: u32 = 16u;
// Complex binomial coefficient C(p, k), k in 1..=16, precomputed on the CPU
// (they depend only on p; see `complex_binomials` in app.rs).
fn cm_coef(k: u32) -> vec2<f32> {
let v = u.cm_coef[(k - 1u) / 2u];
return select(v.xy, v.zw, (k & 1u) == 0u);
}
// Perturbation delta for z -> z^p with a complex p: (Z+e)^p - Z^p.
//
// When |e| << |Z| (the common case: it's the whole reason perturbation
// works), forming Z+e directly would round e away in f32, so instead expand
// = Z^p * ((1+w)^p - 1), w = e/Z, as a Taylor series in w: (1+w)^p - 1 =
// sum_{k=1}^N C(p,k) w^k, with the complex binomial coefficient built up
// incrementally: C(p,k) = C(p,k-1) * (p-(k-1)) / k. Unlike `multibrot_delta`
// (a finite binomial sum for an integer power), this only *converges* — and
// only for |w| < 1 — rather than terminating exactly.
// sum_{k=1}^N C(p,k) w^k. The series stops as soon as the next w^k is
// negligible against the running sum (below f32 precision) — at deep zoom w
// is tiny, so that's typically after 2-3 terms instead of all 16.
//
// Right after a rebase (or near a reference point close to zero, where w is
// singular), e is *not* small relative to Z — that's normal perturbation
@@ -112,13 +123,14 @@ fn complex_multibrot_delta(z: vec2<f32>, e: vec2<f32>, p: vec2<f32>) -> vec2<f32
let w2 = dot(e, e) / dot(z, z);
if w2 < 0.25 {
let w = cdiv(e, z);
var wk = vec2<f32>(1.0, 0.0); // w^0
var coef = vec2<f32>(1.0, 0.0); // C(p,0)
var wk = w; // w^1
var acc = vec2<f32>(0.0, 0.0);
for (var k: u32 = 1u; k <= COMPLEX_MULTIBROT_TERMS; k = k + 1u) {
coef = cdiv(cmul(coef, p - vec2<f32>(f32(k - 1u), 0.0)), vec2<f32>(f32(k), 0.0));
acc = acc + cmul(cm_coef(k), wk);
wk = cmul(wk, w);
acc = acc + cmul(coef, wk);
if dot(wk, wk) < 1e-18 * dot(acc, acc) {
break;
}
}
return cmul(cpow(z, p), acc);
}
@@ -129,7 +141,7 @@ fn complex_multibrot_delta(z: vec2<f32>, e: vec2<f32>, p: vec2<f32>) -> vec2<f32
// where `z` is the reference orbit value X_m. `step_add` (dc) is added by the
// caller. Must match `FractalKind` on the CPU side.
fn advance_delta(z: vec2<f32>, e: vec2<f32>) -> vec2<f32> {
if u.kind == KIND_BURNING_SHIP {
if KIND == KIND_BURNING_SHIP {
// (|x| + i|y|)^2 has real part x^2 - y^2 (an ordinary square delta) and
// imaginary part 2|x y|. The imaginary delta is 2(|x y| - |X Y|); diffabs
// computes it exactly, even where the product x y changes sign — which the
@@ -138,34 +150,34 @@ fn advance_delta(z: vec2<f32>, e: vec2<f32>) -> vec2<f32> {
let base = 2.0 * cmul(z, e) + cmul(e, e);
let dp = z.x * e.y + z.y * e.x + e.x * e.y;
return vec2<f32>(base.x, 2.0 * diffabs(z.x * z.y, dp));
} else if u.kind == KIND_TRICORN {
} else if KIND == KIND_TRICORN {
let cz = conj(z);
let ce = conj(e);
return 2.0 * cmul(cz, ce) + cmul(ce, ce);
} else if u.kind == KIND_MULTIBROT {
} else if KIND == KIND_MULTIBROT {
return multibrot_delta(z, e, clamp(u.power, 2u, 8u));
} else if u.kind == KIND_CELTIC {
} else if KIND == KIND_CELTIC {
// z^2 delta split: sq.x = delta of Re(z^2), sq.y = delta of Im(z^2).
// Celtic abs the real output, so |Re(z^2)| delta = diffabs(Re(Z^2), sq.x).
let sq = 2.0 * cmul(z, e) + cmul(e, e);
return vec2<f32>(diffabs(z.x * z.x - z.y * z.y, sq.x), sq.y);
} else if u.kind == KIND_BUFFALO {
} else if KIND == KIND_BUFFALO {
// Abs both outputs: real |Re(z^2)|, imag -|Im(z^2)| (Im(Z^2) = 2 X Y).
let sq = 2.0 * cmul(z, e) + cmul(e, e);
return vec2<f32>(diffabs(z.x * z.x - z.y * z.y, sq.x),
-diffabs(2.0 * z.x * z.y, sq.y));
} else if u.kind == KIND_PERPENDICULAR {
} else if KIND == KIND_PERPENDICULAR {
// real x^2 - y^2 (ordinary square delta), imag -2 x |y|.
// d(-2 x |y|) = -2[ X·(|Y+ey|-|Y|) + ex·|Y+ey| ]; diffabs gives |Y+ey|-|Y|.
let sq = 2.0 * cmul(z, e) + cmul(e, e);
let da = diffabs(z.y, e.y); // |Y + ey| - |Y|
let abs_yf = abs(z.y) + da; // |Y + ey|
return vec2<f32>(sq.x, -2.0 * (z.x * da + e.x * abs_yf));
} else if u.kind == KIND_LAMBDA {
} else if KIND == KIND_LAMBDA {
// Lambda map: z^{n+1} = λ·z·(1-z). Delta: e = λ·e·(1-2z-e).
let one_minus_2z_minus_e = vec2<f32>(1.0 - 2.0 * z.x - e.x, -2.0 * z.y - e.y);
return cmul(u.lambda_l, cmul(e, one_minus_2z_minus_e));
} else if u.kind == KIND_COMPLEX_MULTIBROT {
} else if KIND == KIND_COMPLEX_MULTIBROT {
return complex_multibrot_delta(z, e, u.complex_power);
}
return 2.0 * cmul(z, e) + cmul(e, e); // Mandelbrot (and Phoenix square part)
@@ -177,17 +189,17 @@ fn advance_delta(z: vec2<f32>, e: vec2<f32>) -> vec2<f32> {
// Burning Ship / Tricorn we use |f'| ~ |2Z|, which keeps the DE magnitude close
// enough to de-speckle filaments.
fn fprime(z: vec2<f32>) -> vec2<f32> {
if u.kind == KIND_MULTIBROT {
if KIND == KIND_MULTIBROT {
let p = clamp(u.power, 2u, 8u);
var zk = vec2<f32>(1.0, 0.0); // Z^0
for (var k: u32 = 1u; k < p; k = k + 1u) {
var zk = z; // Z^1
for (var k: u32 = 2u; k < p; k = k + 1u) {
zk = cmul(zk, z); // -> Z^{p-1}
}
return f32(p) * zk;
} else if u.kind == KIND_LAMBDA {
} else if KIND == KIND_LAMBDA {
// Lambda: f'(z) = λ·(1-2z).
return cmul(u.lambda_l, vec2<f32>(1.0 - 2.0 * z.x, -2.0 * z.y));
} else if u.kind == KIND_COMPLEX_MULTIBROT {
} else if KIND == KIND_COMPLEX_MULTIBROT {
// f'(z) = p * z^(p-1).
return cmul(u.complex_power, cpow(z, u.complex_power - vec2<f32>(1.0, 0.0)));
}
@@ -209,6 +221,10 @@ struct Sample {
// starts at 0); for Julia it is the z-plane offset that seeds the initial delta
// (c is fixed, so nothing is added per step).
fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
// Loop invariants, read once instead of on every iteration.
let max_iter = u.max_iter;
let bailout_sq = u.bailout_sq;
let ref_len = u.ref_len;
let z0 = ref_orbit[0]; // reference start (0 for Mandelbrot, center for Julia)
// Main cardioid / period-2 bulb bypass: those points never escape, so skip
@@ -217,7 +233,7 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
// orbit itself, since X_1 = X_0^2 + C_ref = C_ref. That's only f32-accurate,
// so skip the test once a pixel is smaller than that error (deep zoom),
// where it could misclassify pixels right at the boundary.
if u.kind == KIND_MANDELBROT && u.is_julia == 0u && u.ref_len > 1u && px > 1e-6 {
if KIND == KIND_MANDELBROT && !IS_JULIA && ref_len > 1u && px > 1e-6 {
let c = ref_orbit[1] + offset;
let xq = c.x - 0.25;
let q = xq * xq + c.y * c.y;
@@ -229,95 +245,101 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
}
}
// Set plane: delta starts at 0 and gains dc every step. Julia: the offset
// seeds the delta and nothing is added per step.
var step_add = offset;
var e = vec2<f32>(0.0, 0.0);
// Orbit derivative for distance estimation. For the set plane it is d/dc
// (starts at 0, gains +1 each step); for Julia it is d/dz0 (starts at 1).
var dz = vec2<f32>(0.0, 0.0);
var dz_seed = vec2<f32>(1.0, 0.0);
if IS_JULIA {
step_add = vec2<f32>(0.0, 0.0);
e = offset;
dz = vec2<f32>(1.0, 0.0);
}
// Previous-iterate state for the Phoenix two-term recurrence (delta of
// y_{n-1}, and its derivative for DE). Both start at 0 (y_{-1} = 0).
var e_prev = vec2<f32>(0.0, 0.0);
var dz_prev = vec2<f32>(0.0, 0.0);
if u.is_julia != 0u {
step_add = vec2<f32>(0.0, 0.0);
e = offset;
dz = vec2<f32>(1.0, 0.0);
dz_seed = vec2<f32>(0.0, 0.0);
}
var m: u32 = 0u; // reference index; invariant: y_n = X[m] + e
var m: u32 = 0u; // reference index; invariant: y_n = xm + e, xm = X[m]
var n: u32 = 0u; // total iteration count
var z = vec2<f32>(0.0, 0.0); // full value y_n, kept for coloring
var xm = z0; // X[m], carried so each step loads the orbit once
var z = xm + e; // full value y_n, kept for coloring
var z2 = dot(z, z);
var escaped = false;
loop {
let xm = ref_orbit[m];
z = xm + e;
let z2 = dot(z, z);
if z2 > u.bailout_sq {
escaped = true;
break;
}
if n >= u.max_iter {
break; // interior
}
loop {
if z2 > bailout_sq {
escaped = true;
break;
}
if n >= max_iter {
break; // interior
}
// Propagate the derivative of the full orbit (unaffected by rebasing,
// which only re-expresses the same value). Only when DE is enabled.
// Phoenix's two-term map adds p·dz_{n-1} and carries the previous dz.
if u.de_coloring != 0u {
var dz_new = cmul(fprime(z), dz) + dz_seed;
if u.kind == KIND_PHOENIX {
dz_new = dz_new + cmul(u.phoenix_p, dz_prev);
dz_prev = dz;
}
dz = dz_new;
if DE {
var dz_new = cmul(fprime(z), dz);
if !IS_JULIA {
dz_new.x = dz_new.x + 1.0;
}
if KIND == KIND_PHOENIX {
dz_new = dz_new + cmul(u.phoenix_p, dz_prev);
dz_prev = dz;
}
dz = dz_new;
}
// Advance the delta by this fractal's formula (+ dc for the set plane).
// Phoenix additionally adds p·e_{n-1} and carries the previous delta.
let e_old = e;
e = advance_delta(xm, e) + step_add;
if u.kind == KIND_PHOENIX {
e = e + cmul(u.phoenix_p, e_prev);
e_prev = e_old;
}
m = m + 1u;
n = n + 1u;
let e_old = e;
let z_old = z;
e = advance_delta(xm, e) + step_add;
if KIND == KIND_PHOENIX {
e = e + cmul(u.phoenix_p, e_prev);
e_prev = e_old;
}
m = m + 1u;
n = n + 1u;
// Keep the reference index valid and the delta small.
if m >= u.ref_len {
if m >= ref_len {
// Reference exhausted: any pixel that followed it this far has
// effectively escaped (interior pixels rebase before reaching here).
z = ref_orbit[u.ref_len - 1u] + e;
escaped = true;
break;
}
let y = ref_orbit[m] + e;
if dot(y, y) < dot(e, e) {
// Rebase to index 0: carry the full value as the new delta. Valid
// because y_n = X[0] + (y_n - X[0]); for Mandelbrot X[0]=0.
// Phoenix: after rebasing the implied previous reference is Y[-1]=0,
// so the previous delta becomes the full previous value y_n (= z).
if u.kind == KIND_PHOENIX {
e_prev = z;
}
e = y - z0;
m = 0u;
}
z = xm + e;
escaped = true;
break;
}
xm = ref_orbit[m];
z = xm + e;
z2 = dot(z, z);
if z2 < dot(e, e) {
// Rebase to index 0: carry the full value as the new delta. Valid
// because y_n = X[0] + (y_n - X[0]); for Mandelbrot X[0]=0. The
// full value `z` (and `z2`) is unchanged by the re-expression.
// Phoenix: after rebasing the implied previous reference is Y[-1]=0,
// so the previous delta becomes the full previous value y_{n-1}.
if KIND == KIND_PHOENIX {
e_prev = z_old;
}
e = z - z0;
xm = z0;
m = 0u;
}
}
if !escaped {
return Sample(0.0, 1.0, false); // interior of the set
}
let z2 = dot(z, z);
z2 = dot(z, z);
// Continuous (smooth) iteration count.
let log_zn = 0.5 * log(max(z2, 1.0));
let nu = log2(log_zn / log(2.0));
let nu = log2(log_zn * INV_LN2);
let smooth_i = f32(n) + 1.0 - nu;
// sqrt compresses the huge iteration counts of deep zooms so the palette
@@ -325,7 +347,7 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
let ci = sqrt(max(smooth_i, 0.0));
var de = 1.0;
if u.de_coloring != 0u {
if DE {
// Exterior distance estimate (complex-plane units): |z|·ln|z| / |dz|.
// Divided by the pixel footprint it becomes a distance in pixels; we
// darken toward the boundary (< ~1 px away) so filaments stay crisp
@@ -334,15 +356,15 @@ fn iterate_sample(offset: vec2<f32>, px: f32) -> Sample {
let zmag = sqrt(max(z2, 1.0));
let dzmag = sqrt(max(dot(dz, dz), 1e-20));
let d = zmag * log(zmag) / dzmag;
var max_de = 1.;
if u.shadow != 0u {
max_de = 1000.;
}
let max_de = select(1.0, 1000.0, u.shadow != 0u);
de = clamp(d / max(px, 1e-30), 0.0, max_de);
}
return Sample(ci, de, true);
}
// 1 / ln(2), for the smooth iteration count's log2(ln|z| / ln 2).
const INV_LN2: f32 = 1.4426950408889634;
// Map a sample's escape data through the palette (+ DE darkening). This is the
// only color-dependent step, so it can be redone without re-iterating. Interior
// samples are black.
@@ -353,13 +375,13 @@ fn color_sample(s: Sample) -> vec3<f32> {
return classic_color(s.ci, s.de);
}
// Supersampled escape data at one point: average (ci, DE factor) over the
// AA grid's escaped sub-samples, plus the fraction that landed in the
// interior. Shared by `fs_data` (writes it straight to the data texture) and
// `fs_color`'s shadow branch (used both at the pixel and at its two
// neighbours, to build a DE height field without a texture round-trip).
fn aggregate_sample(base: vec2<f32>, dx: vec2<f32>, dy: vec2<f32>, px: f32) -> vec3<f32> {
let aa = max(u.aa_level, 1u);
// Supersampled escape data at one point: average (ci, DE factor) over an
// `aa`×`aa` grid's escaped sub-samples, plus the fraction that landed in the
// interior. Shared by `fs_data` (1 sample), `fs_refine` (the AA grid, only on
// pixels that need it) and `fs_color`'s shadow branch (used both at the pixel
// and at its two neighbours, to build a DE height field without a texture
// round-trip).
fn aggregate_sample(base: vec2<f32>, dx: vec2<f32>, dy: vec2<f32>, px: f32, aa: u32) -> vec3<f32> {
let inv = 1.0 / f32(aa);
var ci_sum = 0.0;
var de_sum = 0.0;
@@ -386,8 +408,8 @@ fn aggregate_sample(base: vec2<f32>, dx: vec2<f32>, dy: vec2<f32>, px: f32) -> v
// Iteration pass: write per-pixel escape data (color-independent) so a colour
// change is remapped by the cheap colourise pass without re-iterating.
// R = ci (palette parameter), G = DE factor, B = interior fraction (for AA).
// AA is grid-supersampled here; the interior fraction lets the colourise pass
// anti-alias the set boundary (blend toward black) after the fact.
// Always one sample per pixel: anti-aliasing is added afterwards, only where
// it matters, by `fs_refine`.
@fragment
fn fs_data(in: VsOut) -> @location(0) vec4<f32> {
let base = in.centered * u.span + u.dc_offset;
@@ -395,35 +417,82 @@ fn fs_data(in: VsOut) -> @location(0) vec4<f32> {
let dy = dpdy(base);
let px = length(abs(dx) + abs(dy));
return vec4<f32>(aggregate_sample(base, dx, dy, px), 1.0);
return vec4<f32>(aggregate_sample(base, dx, dy, px, 1u), 1.0);
}
// Adaptive-AA thresholds for `fs_refine`: a pixel is supersampled only if a
// 4-neighbour's 1-spp sample differs from its own by more than this. `ci`
// steps are palette-phase steps of `ci * color_scale` (color_scale <= 1 in the
// UI), so 0.02 keeps anything visibly banded; DE is compared relative to its
// own magnitude (it's in pixels, up to 1000 for shadow/3D height fields).
const AA_CI_EPS: f32 = 0.02;
const AA_DE_EPS: f32 = 0.1;
fn aa_differs(c: vec4<f32>, n: vec4<f32>) -> bool {
if c.b != n.b {
return true; // interior / exterior boundary
}
if c.b != 0.0 {
return false; // both interior: uniformly black
}
return abs(n.r - c.r) > AA_CI_EPS || abs(n.g - c.g) > AA_DE_EPS * max(c.g, 0.1);
}
// Adaptive anti-aliasing pass (only run when AA is on): reads `fs_data`'s
// 1-spp texture and re-iterates the full AA grid only for pixels whose
// neighbourhood isn't smooth (set boundary, filaments, palette discontinuities).
// Everywhere else the centre sample already equals the grid average to within
// the thresholds above, so it's copied — which skips the AA cost entirely for
// the interior (the most expensive pixels, each burning max_iter) and for the
// smooth exterior.
@fragment
fn fs_refine(in: VsOut) -> @location(0) vec4<f32> {
// Derivatives first, while control flow is still uniform.
let base = in.centered * u.span + u.dc_offset;
let dx = dpdx(base);
let dy = dpdy(base);
let px = length(abs(dx) + abs(dy));
let p = vec2<i32>(in.pos.xy);
let hi = vec2<i32>(textureDimensions(coarse_tex)) - vec2<i32>(1, 1);
let c = textureLoad(coarse_tex, p, 0);
let l = textureLoad(coarse_tex, max(p - vec2<i32>(1, 0), vec2<i32>(0, 0)), 0);
let r = textureLoad(coarse_tex, min(p + vec2<i32>(1, 0), hi), 0);
let t = textureLoad(coarse_tex, max(p - vec2<i32>(0, 1), vec2<i32>(0, 0)), 0);
let b = textureLoad(coarse_tex, min(p + vec2<i32>(0, 1), hi), 0);
if aa_differs(c, l) || aa_differs(c, r) || aa_differs(c, t) || aa_differs(c, b) {
return vec4<f32>(aggregate_sample(base, dx, dy, px, max(u.aa_level, 1u)), 1.0);
}
return c;
}
// Combined iterate + colour in a single pass, for PNG export (which never needs
// incremental recolouring). The interactive path uses fs_data + the colourise
// pass so colour changes skip iteration.
// incremental recolouring). The interactive path uses fs_data (+ fs_refine) +
// the colourise pass so colour changes skip iteration. Export always runs the
// full AA grid on every pixel, for maximum quality.
@fragment
fn fs_color(in: VsOut) -> @location(0) vec4<f32> {
let base = in.centered * u.span + u.dc_offset;
let dx = dpdx(base);
let dy = dpdy(base);
let px = length(abs(dx) + abs(dy));
let aa = max(u.aa_level, 1u);
if u.shadow != 0u {
// No data texture to sample neighbours from (this pass never runs
// one), so build the same DE height field colorize.wgsl reads from
// the texture by aggregating live, at the pixel and its two
// neighbours a `dx`/`dy` step away.
let here = aggregate_sample(base, dx, dy, px);
let here = aggregate_sample(base, dx, dy, px, aa);
if here.z != 0.0 {
return vec4<f32>(0.1, 0.1, 0.1, 1.0);
}
let right = aggregate_sample(base + dx, dx, dy, px);
let down = aggregate_sample(base + dy, dx, dy, px);
let right = aggregate_sample(base + dx, dx, dy, px, aa);
let down = aggregate_sample(base + dy, dx, dy, px, aa);
let normal = normal_from_heights(here.y, right.y, down.y);
return vec4<f32>(shadow_color(normal), 1.0);
}
let aa = max(u.aa_level, 1u);
let inv = 1.0 / f32(aa);
var acc = vec3<f32>(0.0, 0.0, 0.0);
for (var sy: u32 = 0u; sy < aa; sy = sy + 1u) {