git.lucas.co / cce-ui
GPU-accelerated UI toolkit (Vulkan)
git clone https://git.lucas.co/cce-ui.git

commit9d9364650208ffb6dbb466a43dcc58a7e5a9dfe0
parent8b38c324fd
authorLucas Galante <[email protected]>
date2026-08-14 21:04
refactor: dispatch plate modes by equality; assert the push-constant budget

Two latent traps in the SDF path, neither of which changes a pixel.

The mode dispatch compared RANGES — `> 4.5 && < 5.5`, `< 1.5`, `> 5.5`,
`> 7.5` — so the branch order was load-bearing without saying so. Adding
Prim::Groove, mode 8's arm HAD to precede the fillet arm: appended after, the
`> 5.5` branch would have claimed it, subtracted 4 and drawn every groove as a
ridge. Nothing in the file said so and the compiler could not. Modes are now
named constants compared for equality on a rounded int, so a new mode is inert
wherever it lands rather than silently captured by a neighbour. Every old range
is exactly equivalent on the modes that can reach it, which is why this is a
no-op: verified by pixel comparison, not by reading.

The push-constant block is 32 floats = 128 bytes, which is not "roomy, and
happens to be 128" but EXACTLY maxPushConstantsSize's guaranteed minimum, fully
written. That is the real reason modes 5-8 reinterpret existing fields instead
of adding any. The size was a bare literal in three places; it is one constant
now, with a const assertion that fires if it ever exceeds 128 and says what to
do about it (query the device limit, add a fallback). A runtime check would be
dead code at exactly 128 — it can never fire on a conformant implementation.

Verified: cce-files and a cce-cloud slider popup render BYTE-IDENTICAL before
and after (magick compare AE = 0), covering modes 1, 2, 3, 5, 6/7 and 8.
MODE_RIDGE has no production caller in the workspace, so it is covered by the
equivalence argument only.

Co-Authored-By: Claude <[email protected]>

 src/vk/renderer.rs   | 38 +++++++++++++++++++++++++++++++++-----
 src/vk/shader2d.wgsl | 52 ++++++++++++++++++++++++++++++++++++----------------
 2 files changed, 69 insertions(+), 21 deletions(-)

diff --git a/src/vk/renderer.rs b/src/vk/renderer.rs
index a49e88d..cbbd25a 100644
--- a/src/vk/renderer.rs
+++ b/src/vk/renderer.rs
@@ -42,6 +42,34 @@ pub struct Batch2D {
     pub blur_behind: bool,
 }
 
+/// Floats in the fragment push-constant block: the rounded-rect clip (`rect0`,
+/// `rect1` — 6 clip/flag floats plus the plate mode and corner shape) followed
+/// by [`PlatePush`]'s six vec4s. Field for field, this is shader2d's `RRectClip`.
+pub(crate) const PUSH_CONSTANT_FLOATS: usize = 32;
+
+/// The block in bytes. **This is exactly `maxPushConstantsSize`'s
+/// Vulkan-guaranteed minimum, so the budget is full** — every one of the 32
+/// slots is written. That is why a new SDF mode reinterprets existing fields per
+/// mode (5 reads `p_rect` as centre + radius, 6/7 as centre + radius + wedge
+/// angle, 8 as centre + half-width with `p_radii.xy` a normal) instead of adding
+/// one: there is nothing left to add.
+///
+/// A block over 128 bytes is not portable by construction — 128 is the floor
+/// every conformant implementation must offer, and plenty of drivers offer no
+/// more. So growing this means querying `limits.max_push_constants_size` at
+/// device init and having a real fallback (a uniform buffer, or splitting the
+/// block), not just raising the number. The assertion below is the tripwire: a
+/// runtime check would be dead code today, because at exactly 128 it can never
+/// fire on a conformant device.
+pub(crate) const PUSH_CONSTANT_BYTES: u32 = (PUSH_CONSTANT_FLOATS * 4) as u32;
+
+const _: () = assert!(
+    PUSH_CONSTANT_BYTES <= 128,
+    "the push-constant block has outgrown the 128-byte Vulkan-guaranteed minimum: \
+     query limits.max_push_constants_size at device init and add a fallback path \
+     before raising PUSH_CONSTANT_FLOATS"
+);
+
 /// Push-constant block for one SDF-lit plate batch (physical px throughout).
 /// Mirrors the `p_*` fields of shader2d's `RRectClip`.
 #[derive(Clone, Copy, PartialEq, Debug)]
@@ -532,12 +560,12 @@ impl VkRenderer {
         let set_layouts = [descriptor_set_layout];
         // Push constants: the per-batch rounded-rect clip plus the SDF-lit
         // plate block (eight vec4s, matching shader2d's `RRectClip`), read by
-        // shader2d's fragment stage. 128 bytes — exactly the Vulkan-guaranteed
-        // minimum budget.
+        // shader2d's fragment stage. See `PUSH_CONSTANT_BYTES` — the block is
+        // exactly the Vulkan-guaranteed minimum and completely full.
         let push_ranges = [vk::PushConstantRange::default()
             .stage_flags(vk::ShaderStageFlags::FRAGMENT)
             .offset(0)
-            .size(128)];
+            .size(PUSH_CONSTANT_BYTES)];
         let pipeline_layout = device
             .create_pipeline_layout(
                 &vk::PipelineLayoutCreateInfo::default()
@@ -1870,7 +1898,7 @@ impl VkRenderer {
                             // batch is a plate cover quad.
                             let rr = batch.clip_rrect.unwrap_or([0.0; 5]);
                             let enabled = if batch.clip_rrect.is_some() { 1.0f32 } else { 0.0 };
-                            let mut pc = [0.0f32; 32];
+                            let mut pc = [0.0f32; PUSH_CONSTANT_FLOATS];
                             pc[..5].copy_from_slice(&rr);
                             pc[5] = enabled;
                             pc[7] = clip_shape;
@@ -1938,7 +1966,7 @@ impl VkRenderer {
                     .cmd_bind_vertex_buffers(cmd, 0, &[frame.vertex.buffer], &[0]);
                 // Push constants persist across binds — clear any batch's
                 // rounded clip and plate mode.
-                let pc = [0.0f32; 32];
+                let pc = [0.0f32; PUSH_CONSTANT_FLOATS];
                 self.core.device.cmd_push_constants(
                     cmd,
                     self.pipeline_layout,
diff --git a/src/vk/shader2d.wgsl b/src/vk/shader2d.wgsl
index 927b8f6..05ecdac 100644
--- a/src/vk/shader2d.wgsl
+++ b/src/vk/shader2d.wgsl
@@ -113,6 +113,24 @@ struct RRectClip {
 }
 var<push_constant> rrect_clip: RRectClip;
 
+// Plate modes, as carried in rect1.z (see PlatePush::mode). Compared by EQUALITY
+// on a rounded int, never by range: the ranges these replaced were ordered, and
+// the order was load-bearing without saying so — mode 8's branch had to precede
+// the `> 5.5` fillet branch or the fillet arm would have swallowed it, taken 4
+// off, and drawn every groove as a ridge. Equality makes a new mode inert
+// wherever it is added rather than silently captured by a neighbour.
+const MODE_NONE: i32 = 0;         // not a plate batch
+const MODE_PLATE: i32 = 1;        // raised lit plate: fill + rolled perimeter + CSG carves
+const MODE_RECESS: i32 = 2;       // free carve, interior one step DOWN
+const MODE_BOSS: i32 = 3;         // free carve, interior one step UP
+const MODE_RIDGE: i32 = 4;        // raised rim straddling the boundary
+const MODE_SPHERE: i32 = 5;       // hemisphere-lit disc
+const MODE_FILLET_DOWN: i32 = 6;  // concave inside-corner wall, recessed
+const MODE_FILLET_UP: i32 = 7;    // concave inside-corner wall, raised
+const MODE_GROOVE: i32 = 8;       // slab carve about an arbitrary line
+// Fillet modes rejoin the shared free-carve path as their flat equivalents.
+const FILLET_TO_STEP: i32 = 4;    // 6 -> RECESS, 7 -> BOSS
+
 const TAU: f32 = 6.28318530718;
 // Ambient floor of the plate lighting model: the fraction of illumination that
 // arrives from everywhere rather than from the directional light. Keeps shadow
@@ -279,15 +297,17 @@ fn plate_shade(frag: vec2f, vcol: vec4f) -> vec4f {
     let l = rrect_clip.p_light.xyz;
     let strength = rrect_clip.p_mat.x;
     let flat_shade = PLATE_AMBIENT + (1.0 - PLATE_AMBIENT) * l.z;
+    // The mode arrives as a float only because the push block is all f32.
+    let mode = i32(round(rrect_clip.rect1.z));
 
-    // Mode 5: a sphere-lit disc (the slider thumb). p_rect.xy is the center,
+    // MODE_SPHERE: a sphere-lit disc (the slider thumb). p_rect.xy is the center,
     // p_rect.z the radius, physical px. The disc is shaded as a hemisphere
     // under the same light/material as the plates — ambient floor, diffuse off
     // the sphere normal, the decoupled roll specular (its glint lands where
     // the surface tilt meets the half-vector, ~a third of the way out toward
     // the light) — and, like a plate face, the shade is expressed relative to
     // the flat face so the color at the lit center is exactly the app's.
-    if (rrect_clip.rect1.z > 4.5 && rrect_clip.rect1.z < 5.5) {
+    if (mode == MODE_SPHERE) {
         let c = frag - rrect_clip.p_rect.xy;
         let r = max(rrect_clip.p_rect.z, 0.001);
         let dist = length(c);
@@ -307,7 +327,7 @@ fn plate_shade(frag: vec2f, vcol: vec4f) -> vec4f {
     let d = -gd.z; // positive inside the plate, in px
     let t = max(rrect_clip.p_light.w, 0.001);
 
-    if (rrect_clip.rect1.z < 1.5) {
+    if (mode == MODE_PLATE) {
         let aa = clamp(d + 0.5, 0.0, 1.0); // 1px silhouette anti-aliasing
         if (aa <= 0.0) {
             discard;
@@ -351,8 +371,8 @@ fn plate_shade(frag: vec2f, vcol: vec4f) -> vec4f {
     // e.g. in a widget's own paint): an overlay over whatever is painted
     // beneath — no fill, no silhouette. Junction behavior here is the heuristic
     // host-box fade; grouped features get the exact CSG above.
-    // Mode 2 = recess (interior one step DOWN), mode 3 = boss (interior one
-    // step UP) — the same wall with the height sign flipped. Mode 4 = ridge: a
+    // MODE_RECESS = interior one step DOWN, MODE_BOSS = interior one step UP —
+    // the same wall with the height sign flipped. MODE_RIDGE is a
     // raised bump straddling the boundary, both sides at the base level — ONE
     // profile evaluation, so its crest carries a single specular/shoulder term
     // instead of a boss+recess double-stack.
@@ -360,33 +380,33 @@ fn plate_shade(frag: vec2f, vcol: vec4f) -> vec4f {
     // multiplicative shading (black at alpha 1 - shade); brightening is a
     // translucent white screen.
     //
-    // Concave fillet (mode 6 = recessed, 7 = raised): the wall follows a
+    // Concave fillet (MODE_FILLET_DOWN / MODE_FILLET_UP): the wall follows a
     // quarter ARC whose centre sits out in the pocket — the inside-corner
     // rounding the box SDF cannot express. p_rect.xy = centre, .z = radius;
     // p_radii.x = the wedge's start angle (quarter span, HARD-cut at the
     // tangent lines — the straight walls continue the profile exactly there).
     // Distance/gradient swap to radial; everything downstream is the shared
-    // free-carve path via `eff` (6→2, 7→3).
-    var eff = rrect_clip.rect1.z;
+    // free-carve path via `eff` (minus FILLET_TO_STEP: 6→RECESS, 7→BOSS).
+    var eff = mode;
     var fd = d;
     var fgd = gd.xy;
     var wedge = 1.0;
-    // Groove (mode 8): a SLAB carve — the band of half-width p_rect.z about the
+    // MODE_GROOVE: a SLAB carve — the band of half-width p_rect.z about the
     // line through p_rect.xy with unit normal p_radii.xy. Distance is |signed
     // distance to that line| minus the half-width, so ONE profile evaluation
     // yields both walls (the gradient flips sign across the centre line, tilting
     // them apart) and the groove costs a single specular term. The box SDF is
     // axis-aligned by construction; this is how a mark runs at an angle.
     // Rejoins the shared free-carve path as a recess (eff = 2).
-    if (eff > 7.5) {
+    if (mode == MODE_GROOVE) {
         let nrm = rrect_clip.p_radii.xy;
         let c = frag - rrect_clip.p_rect.xy;
         let s = dot(c, nrm);
         fd = abs(s) - rrect_clip.p_rect.z;
         fgd = nrm * select(-1.0, 1.0, s >= 0.0);
-        eff = 2.0;
-    } else if (eff > 5.5) {
-        eff = eff - 4.0;
+        eff = MODE_RECESS;
+    } else if (mode == MODE_FILLET_DOWN || mode == MODE_FILLET_UP) {
+        eff = mode - FILLET_TO_STEP;
         let c = frag - rrect_clip.p_rect.xy;
         let dist = max(length(c), 1e-4);
         fd = dist - rrect_clip.p_rect.z;
@@ -399,7 +419,7 @@ fn plate_shade(frag: vec2f, vcol: vec4f) -> vec4f {
     let u = clamp(fd / t + 0.5, 0.0, 1.0);
     var slope = 0.0;
     var curv = 0.0;
-    if (eff > 3.5) {
+    if (eff == MODE_RIDGE) {
         // Ridge bump: the carve profile mirrored about the boundary (rising
         // outer half, falling inner half), amplitude halved so the wall tilt
         // matches a step's despite the doubled profile rate.
@@ -412,7 +432,7 @@ fn plate_shade(frag: vec2f, vcol: vec4f) -> vec4f {
         // tints the whole cover quad).
         curv = -rrect_clip.p_mat.w * sin(w * TAU);
     } else {
-        let dir = select(-1.0, 1.0, eff > 2.5);
+        let dir = select(-1.0, 1.0, eff == MODE_BOSS);
         // The profile slope is carve_slope's family: smoothstep-derived
         // normally, smootherstep (zero second derivative at the plateaus)
         // under a continuous-curvature corner_shape — shading eases in and out
@@ -538,7 +558,7 @@ fn fs_main(in: VertexOutput) -> @location(0) vec4f {
     }
 
     // SDF-lit plate batch (mode in the push constants; see plate_shade).
-    if (rrect_clip.rect1.z > 0.5) {
+    if (i32(round(rrect_clip.rect1.z)) != MODE_NONE) {
         let c = plate_shade(in.clip_position.xy, in.color);
         return vec4f(c.rgb, c.a * clip_cov);
     }