diff --git a/Cargo.lock b/Cargo.lock index 51c1c58..15ac94f 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -93,9 +93,9 @@ checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" [[package]] name = "fearless_simd" -version = "0.7.0" +version = "1.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f4beca3cb2444e3304ac30843cc091f44ed58932353cd492ce740067bfce6b12" +checksum = "f3772b63c40606beea8fe1f7b06a60de27ceb9bfc2d9dad3427368ca41dc7e62" [[package]] name = "image" diff --git a/Cargo.toml b/Cargo.toml index a40c2f9..98fbd06 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -25,7 +25,7 @@ categories = ["multimedia::images", "compression"] [workspace.dependencies] j2k-core = "=0.10.0" j2k-metal-support = "=0.10.0" -fearless_simd = "=0.7.0" +fearless_simd = "=1.0.0" cudarc = { version = "=0.19.9", default-features = false, features = [ "std", "driver", diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md index 1919d08..ecdb1e4 100644 --- a/THIRD_PARTY_NOTICES.md +++ b/THIRD_PARTY_NOTICES.md @@ -6,7 +6,7 @@ transcribed from ITU-T T.832. No T.835/JXRLib source code is included. JPEG XR tables added from the ITU-T T.835 / Microsoft JPEG XR reference source must retain the applicable BSD-3-Clause copyright and disclaimer here. -`fearless_simd` 0.7.0 is used through its safe capability-token API and is +`fearless_simd` 1.0.0 is used through its safe capability-token API and is available under MIT or Apache-2.0 licensing. `cudarc` 0.19.9 provides dynamically loaded CUDA Driver API and NVRTC bindings diff --git a/crates/jxr-core/tests/contracts.rs b/crates/jxr-core/tests/contracts.rs index c4ceff3..4c26e05 100644 --- a/crates/jxr-core/tests/contracts.rs +++ b/crates/jxr-core/tests/contracts.rs @@ -115,8 +115,7 @@ fn surface_layout_rejects_overlapping_planar_destinations() { } #[test] -fn core_uses_shared_backend_and_rect_contracts() { - assert_eq!(BackendKind::Cpu, BackendKind::Cpu); +fn shared_rect_accepts_contained_region() { assert!( Rect { x: 1, diff --git a/crates/jxr-native/src/output_format/simd_pack.rs b/crates/jxr-native/src/output_format/simd_pack.rs index e384461..591ab77 100644 --- a/crates/jxr-native/src/output_format/simd_pack.rs +++ b/crates/jxr-native/src/output_format/simd_pack.rs @@ -1,6 +1,6 @@ //! Capability-token dispatch for common unsigned planar packing. -use fearless_simd::{Level, Simd}; +use fearless_simd::{Bytes, Level, Simd, SimdBase, SimdNarrow}; use super::OutputFormatError; @@ -90,28 +90,94 @@ fn pack_accelerated( ) -> bool { #[cfg(any(target_arch = "x86", target_arch = "x86_64"))] if let Some(avx2) = level.as_avx2() { - pack_vectorized(avx2, input, stride, start, dimensions, scaled, output); - return true; + return pack_vectorized(avx2, input, stride, start, dimensions, scaled, output); } #[cfg(target_arch = "aarch64")] if let Some(neon) = level.as_neon() { - pack_vectorized(neon, input, stride, start, dimensions, scaled, output); - return true; + return pack_vectorized(neon, input, stride, start, dimensions, scaled, output); } false } #[inline] fn pack_vectorized( - _simd: S, + simd: S, input: &[i32], stride: usize, start: [usize; 2], dimensions: [usize; 2], scaled: bool, output: &mut [u8], -) { - pack_rows(input, stride, start, dimensions, scaled, output); +) -> bool { + let lanes = S::i32s::LEN; + let packed_lanes = S::u8s::LEN; + let [width, height] = dimensions; + if width < packed_lanes { + return false; + } + + simd.vectorize(|| { + let bias = if scaled { 1_024 } else { 128 }; + let rounding = if scaled { 3 } else { 0 }; + let addition = bias + rounding; + let shift = if scaled { 3 } else { 0 }; + let zero = S::i32s::splat(simd, 0); + let max = S::i32s::splat(simd, 255); + let vector_width = width / packed_lanes * packed_lanes; + for y in 0..height { + let source_start = (start[1] + y) * stride + start[0]; + let source = &input[source_start..source_start + width]; + let destination = &mut output[y * width..(y + 1) * width]; + for (source, destination) in source[..vector_width] + .chunks_exact(packed_lanes) + .zip(destination[..vector_width].chunks_exact_mut(packed_lanes)) + { + let first = pack_lanes(simd, &source[..lanes], addition, shift, zero, max); + let second = + pack_lanes(simd, &source[lanes..2 * lanes], addition, shift, zero, max); + let third = pack_lanes( + simd, + &source[2 * lanes..3 * lanes], + addition, + shift, + zero, + max, + ); + let fourth = pack_lanes(simd, &source[3 * lanes..], addition, shift, zero, max); + first + .saturating_narrow(second) + .saturating_narrow(third.saturating_narrow(fourth)) + .store_slice(destination); + } + for (destination, &sample) in destination[vector_width..] + .iter_mut() + .zip(&source[vector_width..]) + { + *destination = u8::try_from(((sample + bias + rounding) >> shift).clamp(0, 255)) + .expect("sample is clipped to u8"); + } + } + }); + true +} + +#[expect( + clippy::inline_always, + reason = "SIMD helper must inline into the target-feature vectorize context" +)] +#[inline(always)] +fn pack_lanes( + simd: S, + source: &[i32], + addition: i32, + shift: u32, + zero: S::i32s, + max: S::i32s, +) -> S::u32s { + ((S::i32s::from_slice(simd, source) + addition) >> shift) + .max(zero) + .min(max) + .bitcast() } #[inline] @@ -183,4 +249,44 @@ mod tests { assert_eq!(output, expected); } } + + #[test] + fn packing_matches_scalar_at_row_tails_and_clamp_boundaries() { + let width = 37; + let height = 3; + let stride = 41; + let input: Vec = (0..stride * (height + 1)) + .map(|index| match index % 5 { + 0 => -2_000, + 1 => -128, + 2 => 0, + 3 => 1_024, + _ => i32::MAX - 1_027, + }) + .collect(); + for scaled in [false, true] { + let mut output = Vec::new(); + append_u8( + Level::new(), + &input, + stride, + [2, 1], + [width, height], + scaled, + &mut output, + ) + .unwrap(); + let bias = if scaled { 1_024 } else { 128 }; + let rounding = if scaled { 3 } else { 0 }; + let shift = if scaled { 3 } else { 0 }; + let expected: Vec<_> = (1..=height) + .flat_map(|y| &input[y * stride + 2..y * stride + 2 + width]) + .map(|&sample| { + u8::try_from(((sample + bias + rounding) >> shift).clamp(0, 255)) + .expect("sample is clipped to u8") + }) + .collect(); + assert_eq!(output, expected); + } + } } diff --git a/crates/jxr-native/src/reconstruct/simd_dequant.rs b/crates/jxr-native/src/reconstruct/simd_dequant.rs index 0335e0c..dfc995d 100644 --- a/crates/jxr-native/src/reconstruct/simd_dequant.rs +++ b/crates/jxr-native/src/reconstruct/simd_dequant.rs @@ -1,6 +1,6 @@ //! Checked capability-token SIMD for contiguous coefficient scaling. -use fearless_simd::Simd; +use fearless_simd::{Simd, SimdBase}; use jxr_math::quantization::Quantizer; use crate::CpuCapabilities; @@ -49,8 +49,24 @@ pub(super) fn scale_coefficients( } #[inline] -fn scale_vectorized(_simd: S, input: &[i32], output: &mut [i32], step: u32) { - scale_validated(input, output, step); +fn scale_vectorized(simd: S, input: &[i32], output: &mut [i32], step: u32) { + let multiplier = i32::from_ne_bytes(step.to_ne_bytes()); + simd.vectorize(|| { + let lanes = S::i32s::LEN; + let vector_multiplier = S::i32s::splat(simd, multiplier); + let mut input_chunks = input.chunks_exact(lanes); + let mut output_chunks = output.chunks_exact_mut(lanes); + for (source, destination) in input_chunks.by_ref().zip(output_chunks.by_ref()) { + (S::i32s::from_slice(simd, source) * vector_multiplier).store_slice(destination); + } + for (&source, destination) in input_chunks + .remainder() + .iter() + .zip(output_chunks.into_remainder()) + { + *destination = source.wrapping_mul(multiplier); + } + }); } #[inline] @@ -99,4 +115,17 @@ mod tests { .is_err() ); } + + #[test] + fn coefficient_scaling_preserves_negative_values_and_tail() { + let input: Vec<_> = (-34..35).collect(); + let quantizer = Quantizer::new(113).unwrap(); + let mut output = vec![0; input.len()]; + scale_coefficients(CpuCapabilities::detect(), quantizer, &input, &mut output).unwrap(); + let expected: Vec<_> = input + .iter() + .map(|&value| quantizer.dequantize(value).unwrap()) + .collect(); + assert_eq!(output, expected); + } } diff --git a/crates/jxr/fuzz/Cargo.lock b/crates/jxr/fuzz/Cargo.lock index 9be014c..faf8f55 100644 --- a/crates/jxr/fuzz/Cargo.lock +++ b/crates/jxr/fuzz/Cargo.lock @@ -68,9 +68,9 @@ checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" [[package]] name = "fearless_simd" -version = "0.7.0" +version = "1.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f4beca3cb2444e3304ac30843cc091f44ed58932353cd492ce740067bfce6b12" +checksum = "f3772b63c40606beea8fe1f7b06a60de27ceb9bfc2d9dad3427368ca41dc7e62" [[package]] name = "find-msvc-tools"