diff --git a/vello_cpu/src/fine/common/image.rs b/vello_cpu/src/fine/common/image.rs index 1095ab7f1..fd6732c72 100644 --- a/vello_cpu/src/fine/common/image.rs +++ b/vello_cpu/src/fine/common/image.rs @@ -4,12 +4,119 @@ use crate::fine::macros::{f32x16_painter, u8x16_painter}; use crate::fine::{PosExt, Splat4thExt, u8_to_f32}; use crate::kurbo::Point; +use crate::util::scalar::div_255; +use alloc::vec::Vec; use vello_common::encode::EncodedImage; use vello_common::fearless_simd::{ Bytes, Select, Simd, SimdBase, SimdFloat, f32x4, f32x16, u8x16, u32x4, }; use vello_common::pixmap::Pixmap; use vello_common::simd::element_wise_splat; +use vello_common::tile::Tile; +use vello_common::util::{narrow, normalized_mul_u8}; + +/// Gather the alpha channel of `pixmap` for a tile-high span of `width` pixel columns +/// whose top-left pixel samples the image at `(x, y)`, clamping out-of-bounds coordinates +/// to the image edges (i.e. `Extend::Pad`). +/// +/// The result is written into the first `width * Tile::HEIGHT` elements of `out` (which +/// is grown if necessary) in column-major order, like strip alphas. If `alphas` is provided, +/// each sample is additionally multiplied by the corresponding coverage. +#[inline(always)] +pub(crate) fn pad_alpha_mask( + simd: S, + pixmap: &Pixmap, + x: i32, + y: i32, + width: usize, + alphas: Option<&[u8]>, + out: &mut Vec, +) { + simd.vectorize( + #[inline(always)] + || { + let img_width = i32::from(pixmap.width()); + let img_height = i32::from(pixmap.height()); + debug_assert!(img_width > 0 && img_height > 0, "image must not be empty"); + let data = pixmap.data(); + let [r0, r1, r2, r3]: [usize; Tile::HEIGHT as usize] = core::array::from_fn(|row| { + (y + row as i32).clamp(0, img_height - 1) as usize * img_width as usize + }); + + let len = width * Tile::HEIGHT as usize; + if out.len() < len { + out.resize(len, 0); + } + let out = &mut out[..len]; + + if cfg!(target_endian = "little") && x >= 0 && x + width as i32 <= img_width { + // Fast path: All columns are in bounds, so we can load 4 consecutive pixels + // of each row at once and transpose their alphas into column-major order. + const CHUNK: usize = 4 * Tile::HEIGHT as usize; + let bytes = pixmap.data_as_u8_slice(); + let x = x as usize; + let load = |row_start: usize, col: usize| { + let start = (row_start + x + col) * 4; + u32x4::from_bytes(u8x16::from_slice(simd, &bytes[start..start + 16])) + }; + let mut chunks = out.chunks_exact_mut(CHUNK); + + for (chunk_idx, chunk) in chunks.by_ref().enumerate() { + let col = chunk_idx * 4; + // With little endian, the alpha channel is stored in the most significant + // byte of each pixel. + let packed = (load(r0, col) >> 24) + | ((load(r1, col) >> 24) << 8) + | ((load(r2, col) >> 24) << 16) + | ((load(r3, col) >> 24) << 24); + let mut packed = packed.to_bytes(); + if let Some(alphas) = alphas { + let alphas = u8x16::from_slice(simd, &alphas[col * 4..col * 4 + CHUNK]); + packed = narrow(normalized_mul_u8(packed, alphas)); + } + packed.store_slice(chunk); + } + + let remainder = chunks.into_remainder(); + let col_offset = (width / 4) * 4; + for (col, column) in remainder + .chunks_exact_mut(Tile::HEIGHT as usize) + .enumerate() + { + let img_x = x + col_offset + col; + for (row, (dst, row_start)) in + column.iter_mut().zip([r0, r1, r2, r3]).enumerate() + { + let alpha = data[row_start + img_x].a; + *dst = match alphas { + Some(alphas) => div_255( + u16::from(alpha) * u16::from(alphas[(col_offset + col) * 4 + row]), + ) as u8, + None => alpha, + }; + } + } + } else { + let columns = out.chunks_exact_mut(Tile::HEIGHT as usize); + for (col, column) in columns.enumerate() { + let img_x = (x + col as i32).clamp(0, img_width - 1) as usize; + column.copy_from_slice(&[ + data[r0 + img_x].a, + data[r1 + img_x].a, + data[r2 + img_x].a, + data[r3 + img_x].a, + ]); + } + + if let Some(alphas) = alphas { + for (dst, alpha) in out.iter_mut().zip(alphas) { + *dst = div_255(u16::from(*dst) * u16::from(*alpha)) as u8; + } + } + } + }, + ); +} /// A painter for nearest-neighbor images with no skewing. #[derive(Debug)] @@ -547,6 +654,55 @@ mod tests { use super::*; use vello_common::fearless_simd::{Fallback, SimdMask}; + #[test] + fn pad_alpha_mask_matches_reference() { + let simd = Fallback::new(); + let (width, height) = (9_u16, 7_u16); + let mut pixmap = Pixmap::new(width, height); + for y in 0..height { + for x in 0..width { + let a = (y * width + x) as u8 * 3; + pixmap.set_pixel( + x, + y, + vello_common::color::PremulRgba8 { + r: 0, + g: 0, + b: 0, + a, + }, + ); + } + } + let alphas: Vec = (0..64).map(|i| (i * 37 % 256) as u8).collect(); + + for (x, y) in [(0, 0), (1, 2), (5, 3), (-3, -2), (6, 5), (-5, 6)] { + for span_width in [4_usize, 8, 16] { + for alphas in [None, Some(&alphas[..span_width * 4])] { + let mut out = Vec::new(); + pad_alpha_mask(simd, &pixmap, x, y, span_width, alphas, &mut out); + + for col in 0..span_width { + for row in 0..4 { + let img_x = (x + col as i32).clamp(0, i32::from(width) - 1) as u16; + let img_y = (y + row as i32).clamp(0, i32::from(height) - 1) as u16; + let idx = col * 4 + row; + let mut expected = pixmap.sample(img_x, img_y).a; + if let Some(alphas) = alphas { + expected = + div_255(u16::from(expected) * u16::from(alphas[idx])) as u8; + } + assert_eq!( + out[idx], expected, + "mismatch at x={x}, y={y}, width={span_width}, col={col}, row={row}" + ); + } + } + } + } + } + } + fn assert_extend(mode: crate::peniko::Extend, max: f32, values: [f32; 4], expected: [u32; 4]) { let simd = Fallback::new(); let inv_max = f32x4::splat(simd, 1.0 / max); diff --git a/vello_cpu/src/fine/mod.rs b/vello_cpu/src/fine/mod.rs index fdbf32b48..0fcd469be 100644 --- a/vello_cpu/src/fine/mod.rs +++ b/vello_cpu/src/fine/mod.rs @@ -19,9 +19,11 @@ pub(crate) use crate::fine::common::gradient::calculate_t_vals; pub(crate) use crate::fine::common::gradient::linear::SimdLinearKind; pub(crate) use crate::fine::common::gradient::radial::SimdRadialKind; pub(crate) use crate::fine::common::gradient::sweep::SimdSweepKind; -use crate::fine::common::image::{FilteredImagePainter, NNImagePainter, PlainNNImagePainter}; +use crate::fine::common::image::{ + FilteredImagePainter, NNImagePainter, PlainNNImagePainter, pad_alpha_mask, +}; use crate::fine::common::rounded_blurred_rect::BlurredRoundedRectFiller; -use crate::peniko::{BlendMode, ImageQuality}; +use crate::peniko::{BlendMode, Extend, ImageQuality}; use crate::region::Region; use crate::util::EncodedImageExt; use alloc::vec; @@ -29,7 +31,7 @@ use alloc::vec::Vec; use core::fmt::Debug; use core::iter; use vello_common::TargetInit; -use vello_common::color::AlphaColor; +use vello_common::color::{AlphaColor, Srgb}; use vello_common::encode::{ EncodedBlurredRoundedRectangle, EncodedGradient, EncodedImage, EncodedKind, EncodedPaint, }; @@ -40,7 +42,7 @@ use vello_common::fearless_simd::{ use vello_common::filter_effects::Filter; use vello_common::kurbo::Affine; use vello_common::mask::Mask; -use vello_common::paint::{ImageResolver, ImageSource, Paint, PremulColor, Tint}; +use vello_common::paint::{ImageResolver, ImageSource, Paint, PremulColor, Tint, TintMode}; use vello_common::pixmap::Pixmap; use vello_common::simd::Splat4thExt; use vello_common::tile::Tile; @@ -544,6 +546,11 @@ pub struct Fine> { paint_buf: Vec, /// Buffer for storing gradient interpolation parameters (t values). f32_buf: Vec, + /// Buffer for storing per-pixel coverage of alpha-mask image fills. + mask_buf: Vec, + /// The most recently used tint color of an alpha-mask image fill, and its + /// premultiplied and converted representation. + alpha_mask_color: Option<(AlphaColor, [T::Numeric; 4])>, /// The current strip row y-coordinate in scene/filter coordinates. row_y: u16, /// The origin of the current target we are rendering into. @@ -564,6 +571,8 @@ impl> Fine { buffer_pool: VecPool::new(false), paint_buf: Vec::new(), f32_buf: Vec::new(), + mask_buf: Vec::new(), + alpha_mask_color: None, row_y: 0, origin: (0, 0), } @@ -1089,6 +1098,55 @@ impl> Fine { }; let tint = image.tint.as_ref(); + // Fast path for alpha masks (e.g. glyphs from the glyph atlas) that are + // drawn pixel-aligned: Instead of sampling the image, tinting it and then + // compositing the result, use the image alpha as additional coverage + // for a solid fill with the tint color. + if let Some(tint) = tint + && tint.mode == TintMode::AlphaMask + && image.sampler.x_extend == Extend::Pad + && image.sampler.y_extend == Extend::Pad + && pixmap.width() > 0 + && pixmap.height() > 0 + && let Some((dx, dy)) = image.integer_translation() + { + pad_alpha_mask( + simd, + &pixmap, + i32::from(sample_x) + dx, + i32::from(sample_y) + dy, + width, + alphas, + &mut self.mask_buf, + ); + let mask = Some(&self.mask_buf[..width * Tile::HEIGHT as usize]); + let color = match self.alpha_mask_color { + Some((cached, color)) if cached == tint.color => color, + _ => { + let color = T::extract_color(PremulColor::from_alpha_color(tint.color)); + self.alpha_mask_color = Some((tint.color, color)); + color + } + }; + + if default_blend && attrs.mask.is_none() { + T::alpha_composite_solid(simd, dest, color, mask); + } else { + T::blend( + simd, + dest, + x, + y, + iter::repeat(T::Composite::from_color(simd, color)), + attrs.blend_mode, + mask, + attrs.mask.as_ref(), + ); + } + + return; + } + match (image.has_skew(), image.nearest_neighbor()) { (false, false) => { // Axis-aligned with filtering - use optimized plain painters diff --git a/vello_cpu/src/util.rs b/vello_cpu/src/util.rs index e1daaaa89..aa89ebd09 100644 --- a/vello_cpu/src/util.rs +++ b/vello_cpu/src/util.rs @@ -77,6 +77,7 @@ impl NormalizedMulExt for u8x32 { pub(crate) trait EncodedImageExt { fn has_skew(&self) -> bool; fn nearest_neighbor(&self) -> bool; + fn integer_translation(&self) -> Option<(i32, i32)>; } impl EncodedImageExt for EncodedImage { @@ -87,6 +88,35 @@ impl EncodedImageExt for EncodedImage { fn nearest_neighbor(&self) -> bool { self.sampler.quality == ImageQuality::Low } + + /// If the image is sampled with nearest-neighbor filtering and its transform is a + /// pure integer translation, return the offset that maps a pixel position to + /// the image pixel it samples. + #[inline] + fn integer_translation(&self) -> Option<(i32, i32)> { + // Beyond this magnitude, the `f32` math of the image painters might no longer + // be exact, so don't bother. + const MAX_OFFSET: f64 = (1 << 20) as f64; + + // Note that we purposefully avoid `f64::round`, since it is a libm call on + // some targets and this is called for each command. + let to_integer = |v: f64| { + if v.is_nan() || v.abs() >= MAX_OFFSET { + return None; + } + let rounded = if v >= 0.0 { v + 0.5 } else { v - 0.5 } as i32; + ((v - f64::from(rounded)) as f32) + .is_nearly_zero() + .then_some(rounded) + }; + + let [a, b, c, d, e, f] = self.transform.as_coeffs(); + if !self.nearest_neighbor() || a != 1.0 || b != 0.0 || c != 0.0 || d != 1.0 { + return None; + } + + Some((to_integer(e)?, to_integer(f)?)) + } } pub(crate) trait Premultiply {