diff --git a/CHANGELOG.md b/CHANGELOG.md index 8c644b2e0..23f3ace14 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -13,6 +13,33 @@ and stale roadmap entries have been removed from the public documentation set. - Improves grouped JPEG Metal decode, pooled output reuse, HT cleanup, and inverse wavelet transform dispatch with regression coverage. +- JPEG DCT-scaled decodes (1/2, 1/4, 1/8) now match libjpeg-turbo, and + therefore OpenSlide's NDPI/VMS levels, bit for bit. Checked against a + committed libjpeg-turbo 3.1.4.1 reference matrix covering every chroma + layout, restart intervals, progressive scans and 12-bit precision: + - Reduced IDCTs round like libjpeg's `DESCALE` instead of truncating. + - Each component gets libjpeg-turbo's IDCT size: 4:2:0 and 4:1:0 chroma is + decoded with a larger reduced IDCT, so 4:2:0 needs no upsampling below + full size. + - Chroma is replicated instead of smoothed at 1/8 scale, and at any scale + (including full size) when 2:1 chroma is at most two samples wide. + - Progressive images apply the reduced IDCT to their coefficients instead + of decimating a full-size decode. + - 12-bit images use 12-bit reduced IDCTs instead of decimating a full-size + decode. 12-bit 4:4:0, 4:1:1, 4:1:0 and 1x4 layouts still return + `NotImplemented`. +- Fixes a panic in 12-bit region decodes when the region excluded image + blocks on its left or right. +- Includes output rows in the 12-bit scratch budget so tall, narrow JPEGs + decode without false memory-cap failures. +- `j2k-jpeg-cuda` CPU-backed scaled and region-scaled surfaces now have the + scaled dimensions; they were sized from the source rectangle. +- JPEG Metal scaled decodes follow the same rules, and 4:2:0/4:2:2 region + decodes keep the neighbouring chroma their smoothing reads when a region + ends inside the image. JPEG images at most four pixels wide with 4:2:0 or + 4:2:2 chroma are no longer Metal or CUDA fast shapes: explicit Metal + requests reject them and automatic routing decodes them on the CPU. + - Upgrades `fearless_simd` to 1.0.0 for the `j2k-jpeg` AArch64 NEON kernels and the optional `j2k-native` `simd` feature. The dependency is private, so public APIs are unchanged. `j2k-jpeg` keeps its private exact-AVX2 token on diff --git a/corpus/conformance/baseline_444_restart_67x45.jpg b/corpus/conformance/baseline_444_restart_67x45.jpg new file mode 100644 index 000000000..88a73109e Binary files /dev/null and b/corpus/conformance/baseline_444_restart_67x45.jpg differ diff --git a/corpus/conformance/baseline_444_restart_67x45_eighth.rgb b/corpus/conformance/baseline_444_restart_67x45_eighth.rgb new file mode 100644 index 000000000..8b7ad6f25 Binary files /dev/null and b/corpus/conformance/baseline_444_restart_67x45_eighth.rgb differ diff --git a/corpus/conformance/baseline_444_restart_67x45_half.rgb b/corpus/conformance/baseline_444_restart_67x45_half.rgb new file mode 100644 index 000000000..45c607d51 Binary files /dev/null and b/corpus/conformance/baseline_444_restart_67x45_half.rgb differ diff --git a/corpus/conformance/baseline_444_restart_67x45_quarter.rgb b/corpus/conformance/baseline_444_restart_67x45_quarter.rgb new file mode 100644 index 000000000..4d552de32 Binary files /dev/null and b/corpus/conformance/baseline_444_restart_67x45_quarter.rgb differ diff --git a/corpus/conformance/generate.sh b/corpus/conformance/generate.sh index a24a88082..b8f85c68d 100755 --- a/corpus/conformance/generate.sh +++ b/corpus/conformance/generate.sh @@ -54,6 +54,48 @@ sys.stdout.buffer.write(header + body) PY } +# Integer-only texture (no floating point or library RNG) so every host +# produces the same pixels: triangle waves, two flat patches that code as +# DC-only blocks, LCG noise, and a saturated corner. +write_texture_ppm() { + local width="$1" + local height="$2" + python3 - "$width" "$height" <<'PY' +import sys + +width = int(sys.argv[1]) +height = int(sys.argv[2]) +state = 20260927 + + +def noise(): + global state + state = (state * 1103515245 + 12345) & 0x7FFFFFFF + return state % 121 - 60 + + +header = f"P6\n{width} {height}\n255\n".encode() +body = bytearray() +for y in range(height): + for x in range(width): + rgb = [ + 28 + abs((x * 9 + y * 4) % 144 - 72) * 200 // 72, + 40 + x * 180 // max(width - 1, 1), + 200 - y * 150 // max(height - 1, 1) + abs((x + y) * 7 % 64 - 32) - 16, + ] + if y < 16 and 40 <= x < 56: + rgb = [201, 77, 143] + if 24 <= y < 40 and 8 <= x < 24: + rgb = [19, 233, 98] + if y >= 30 and x >= 40: + rgb = [value + noise() for value in rgb] + if y < 10 and x < 12: + rgb = [255, 255, 0] + body.extend(min(max(value, 0), 255) for value in rgb) +sys.stdout.buffer.write(header + bytes(body)) +PY +} + strip_pnm_header() { python3 -c 'import sys data = sys.stdin.buffer.read() @@ -113,6 +155,20 @@ write_gray_pgm 8 8 \ djpeg -grayscale grayscale_8x8.jpg | strip_pnm_header > grayscale_8x8.gray assert_size grayscale_8x8.gray 64 +# DCT-scaled references: the reduced IDCTs behind OpenSlide's NDPI/VMS levels. +write_texture_ppm 67 45 \ + | cjpeg -quality 90 -sample 1x1,1x1,1x1 -baseline -optimize -restart 3B \ + -outfile baseline_444_restart_67x45.jpg +djpeg -rgb -scale 1/2 baseline_444_restart_67x45.jpg \ + | strip_pnm_header > baseline_444_restart_67x45_half.rgb +assert_size baseline_444_restart_67x45_half.rgb 2346 +djpeg -rgb -scale 1/4 baseline_444_restart_67x45.jpg \ + | strip_pnm_header > baseline_444_restart_67x45_quarter.rgb +assert_size baseline_444_restart_67x45_quarter.rgb 612 +djpeg -rgb -scale 1/8 baseline_444_restart_67x45.jpg \ + | strip_pnm_header > baseline_444_restart_67x45_eighth.rgb +assert_size baseline_444_restart_67x45_eighth.rgb 162 + cat > manifest.json < manifest.json < cjpeg -sample argument (luma factors; chroma are 1x1), or None for +# grayscale. +LAYOUTS = [ + ("gray", None), + ("444", "1x1"), + ("422", "2x1"), + ("440", "1x2"), + ("420", "2x2"), + ("411", "4x1"), + ("410", "4x2"), + ("1x4", "1x4"), +] + +# 3 and 4 pixel wide images leave 2:1 chroma at most 2 samples wide at full +# size; 9 does at 1/4; the larger sizes span several MCUs and partial edges. +SIZES_8BIT = [(3, 3), (4, 4), (9, 9), (18, 14), (45, 31), (67, 45)] +# Tall narrow images make the component planes dominate the scratch budget, +# while full-size 2:1 chroma still takes the scaled (replicating) writer. +SIZES_12BIT = [(4, 4), (9, 9), (45, 31), (4, 64)] + +CODINGS_8BIT = [ + ("baseline", ["-baseline", "-optimize"]), + ("restart", ["-baseline", "-optimize", "-restart", "1B"]), + ("progressive", ["-progressive"]), +] +CODINGS_12BIT = [ + ("extended", ["-optimize"]), + ("progressive", ["-progressive"]), +] + +SCALES = [1, 2, 4, 8] + + +def texture(width, height, precision): + """Deterministic integer-only texture with strong chroma detail.""" + state = 20260927 + maxval = (1 << precision) - 1 + shift = precision - 8 + pixels = [] + for y in range(height): + for x in range(width): + rgb = [ + 28 + abs((x * 9 + y * 4) % 144 - 72) * 200 // 72, + 40 + x * 180 // max(width - 1, 1), + 200 - y * 150 // max(height - 1, 1) + abs((x + y) * 7 % 64 - 32) - 16, + ] + if (x + 2 * y) % 5 == 0: + rgb = [rgb[2], rgb[0], rgb[1]] + if y >= height // 2 and x >= width // 2: + state = (state * 1103515245 + 12345) & 0x7FFFFFFF + rgb = [value + state % 121 - 60 for value in rgb] + # Saturated corner, kept off tiny images so they stay textured. + if width >= 9 and y < 3 and x < 3: + rgb = [255, 255, 0] + rgb = [min(max(value, 0), 255) for value in rgb] + if shift: + rgb = [(value << shift) | ((x * 5 + y * 3 + i) % (1 << shift)) for i, value in enumerate(rgb)] + rgb = [min(value, maxval) for value in rgb] + pixels.append(rgb) + return pixels + + +def pnm(width, height, precision, gray): + maxval = (1 << precision) - 1 + magic = "P5" if gray else "P6" + body = bytearray() + for rgb in texture(width, height, precision): + samples = [(rgb[0] * 77 + rgb[1] * 150 + rgb[2] * 29) >> 8] if gray else rgb + for value in samples: + if maxval > 255: + body += value.to_bytes(2, "big") + else: + body.append(value) + return f"{magic}\n{width} {height}\n{maxval}\n".encode() + bytes(body) + + +def strip_pnm_header(data): + index = 0 + for _ in range(3): + index = data.index(b"\n", index) + 1 + return data[index:] + + +def run(args, stdin): + return subprocess.run(args, input=stdin, capture_output=True, check=True).stdout + + +def main(): + probe = subprocess.run(["cjpeg", "-version"], capture_output=True, check=False) + version = (probe.stderr or probe.stdout).decode().strip().splitlines()[0] + if "libjpeg-turbo version 3." not in version: + sys.exit(f"error: need libjpeg-turbo 3.x cjpeg/djpeg, found {version!r}") + + os.makedirs(OUT_DIR, exist_ok=True) + data = bytearray() + lines = [ + f"# Generated by corpus/conformance/scaled_matrix.py with {version}", + "# name\tlayout\tcoding\tprecision\twidth\theight\tjpeg\tscale1\tscale2\tscale4\tscale8", + ] + + def append(blob): + start = len(data) + data.extend(blob) + return f"{start}:{len(blob)}" + + matrix = [(8, SIZES_8BIT, CODINGS_8BIT), (12, SIZES_12BIT, CODINGS_12BIT)] + for precision, sizes, codings in matrix: + for layout, sample in LAYOUTS: + gray = sample is None + for width, height in sizes: + for coding, coding_args in codings: + args = ["cjpeg", "-quality", "90", "-precision", str(precision)] + args += ["-grayscale"] if gray else ["-sample", f"{sample},1x1,1x1"] + jpeg = run(args + coding_args, pnm(width, height, precision, gray)) + refs = [] + for denom in SCALES: + out = ["djpeg", "-dct", "int", "-scale", f"1/{denom}"] + out += ["-grayscale"] if gray else ["-rgb"] + ref = strip_pnm_header(run(out, jpeg)) + channels = 1 if gray else 3 + bytes_per_sample = 2 if precision > 8 else 1 + expected = ( + -(-width // denom) * -(-height // denom) * channels * bytes_per_sample + ) + if len(ref) != expected: + sys.exit(f"error: {layout} {width}x{height} {coding} 1/{denom}: {len(ref)} != {expected}") + refs.append(append(ref)) + name = f"{layout}_{width}x{height}_{coding}_{precision}bit" + lines.append( + "\t".join( + [name, layout, coding, str(precision), str(width), str(height), append(jpeg)] + + refs + ) + ) + + with open(os.path.join(OUT_DIR, "data.bin"), "wb") as handle: + handle.write(data) + with open(os.path.join(OUT_DIR, "cases.tsv"), "w", encoding="utf-8") as handle: + handle.write("\n".join(lines) + "\n") + print(f"wrote {len(lines) - 2} cases, {len(data)} bytes") + + +if __name__ == "__main__": + main() diff --git a/crates/j2k-jpeg-cuda/src/decoder.rs b/crates/j2k-jpeg-cuda/src/decoder.rs index a49dad67b..6fba8e258 100644 --- a/crates/j2k-jpeg-cuda/src/decoder.rs +++ b/crates/j2k-jpeg-cuda/src/decoder.rs @@ -143,16 +143,17 @@ impl<'a> Decoder<'a> { ), )); } - let (bytes, outcome) = self + let (bytes, _) = self .inner .decode_request(DecodeRequest::scaled(fmt, scale))?; - wrap_surface( - bytes, - (outcome.decoded.w, outcome.decoded.h), - fmt, - backend, - session, - ) + // `DecodeOutcome::decoded` is the source rectangle; the surface holds + // the scaled image. + let (width, height) = self.inner.info().dimensions; + let dims = ( + width.div_ceil(scale.denominator()), + height.div_ceil(scale.denominator()), + ); + wrap_surface(bytes, dims, fmt, backend, session) } fn decode_region_scaled_to_surface_impl( @@ -171,16 +172,13 @@ impl<'a> Decoder<'a> { ), )); } - let (bytes, outcome) = + let (bytes, _) = self.inner .decode_request(DecodeRequest::region_scaled(fmt, roi.into(), scale))?; - wrap_surface( - bytes, - (outcome.decoded.w, outcome.decoded.h), - fmt, - backend, - session, - ) + // `DecodeOutcome::decoded` is the source ROI; the surface holds its + // scaled covering rectangle. + let scaled = roi.scaled_covering(scale); + wrap_surface(bytes, (scaled.w, scaled.h), fmt, backend, session) } } diff --git a/crates/j2k-jpeg-cuda/tests/host_surface/regions.rs b/crates/j2k-jpeg-cuda/tests/host_surface/regions.rs index 77be676a7..ed0b3e90b 100644 --- a/crates/j2k-jpeg-cuda/tests/host_surface/regions.rs +++ b/crates/j2k-jpeg-cuda/tests/host_surface/regions.rs @@ -42,6 +42,24 @@ fn auto_region_scaled_surface_matches_host_decode() { assert_eq!(surface.as_host_bytes(), Some(host.as_slice())); } +#[test] +fn auto_scaled_surface_has_scaled_dimensions_and_matches_host_decode() { + let scale = Downscale::Quarter; + let mut decoder = Decoder::new(BASELINE_420).expect("decoder"); + let surface = decoder + .decode_scaled_to_device(PixelFormat::Rgb8, scale, BackendRequest::Auto) + .expect("surface"); + assert_eq!(surface.backend_kind(), j2k_core::BackendKind::Cpu); + // The 16x16 fixture at 1/4 is 4x4, not the source size. + assert_eq!(surface.dimensions(), (4, 4)); + + let (host, _) = j2k_jpeg::Decoder::new(BASELINE_420) + .expect("host decoder") + .decode_request(DecodeRequest::scaled(PixelFormat::Rgb8, scale)) + .expect("host decode"); + assert_eq!(surface.as_host_bytes(), Some(host.as_slice())); +} + #[test] fn tile_batch_region_scaled_auto_surface_matches_host_decode() { let roi = Rect { diff --git a/crates/j2k-jpeg-metal/src/abi.rs b/crates/j2k-jpeg-metal/src/abi.rs index 1a05d4d62..d1cd324f8 100644 --- a/crates/j2k-jpeg-metal/src/abi.rs +++ b/crates/j2k-jpeg-metal/src/abi.rs @@ -128,6 +128,15 @@ pub(crate) struct JpegBaselineEntropyEncodeBatchJob<'a> { pub(crate) entropy_capacity: usize, } +/// Pack-kernel chroma modes (`CHROMA_*` in `shaders_shared.metal`), chosen as +/// libjpeg-turbo chooses its upsampler for the decode scale. +pub(crate) const CHROMA_FANCY: u32 = 0; +/// Each chroma sample repeated: smoothing is off at 1/8 scale and for chroma +/// at most two samples wide. +pub(crate) const CHROMA_REPLICATE: u32 = 1; +/// Chroma planes already at output resolution: 4:2:0 decoded below full size. +pub(crate) const CHROMA_UNSAMPLED: u32 = 2; + #[cfg(target_os = "macos")] #[repr(C)] #[derive(Clone, Copy)] @@ -147,6 +156,8 @@ pub(crate) struct JpegFast420Params { pub(crate) out_format: u32, pub(crate) origin_x: u32, pub(crate) origin_y: u32, + /// Chroma upsampling mode (`CHROMA_*` in the shaders). + pub(crate) chroma_mode: u32, } #[cfg(target_os = "macos")] @@ -216,6 +227,8 @@ pub(crate) struct JpegFast420WindowedPackParams { pub(crate) out_stride: u32, pub(crate) alpha: u32, pub(crate) out_format: u32, + /// Chroma upsampling mode (`CHROMA_*` in the shaders). + pub(crate) chroma_mode: u32, } #[cfg(target_os = "macos")] @@ -312,6 +325,8 @@ pub(crate) struct JpegWindowedPackBatchParams { pub(crate) alpha: u32, pub(crate) mode: u32, pub(crate) out_format: u32, + /// Chroma upsampling mode (`CHROMA_*` in the shaders). + pub(crate) chroma_mode: u32, } #[cfg(target_os = "macos")] @@ -328,6 +343,8 @@ pub(crate) struct JpegWindowedTexturePackBatchParams { pub(crate) height: u32, pub(crate) tile_index: u32, pub(crate) alpha: u32, + /// Chroma upsampling mode (`CHROMA_*` in the shaders). + pub(crate) chroma_mode: u32, } #[cfg(target_os = "macos")] @@ -598,6 +615,7 @@ impl_gpu_readback_abi!( out_format: u32, origin_x: u32, origin_y: u32, + chroma_mode: u32, }, JpegFast420ScaledParams { scaled_width: u32, @@ -651,6 +669,7 @@ impl_gpu_readback_abi!( out_stride: u32, alpha: u32, out_format: u32, + chroma_mode: u32, }, JpegFast420BatchParams { width: u32, @@ -712,6 +731,7 @@ impl_gpu_readback_abi!( alpha: u32, mode: u32, out_format: u32, + chroma_mode: u32, }, JpegWindowedTexturePackBatchParams { src_width: u32, @@ -724,6 +744,7 @@ impl_gpu_readback_abi!( height: u32, tile_index: u32, alpha: u32, + chroma_mode: u32, }, JpegTexturePackBatchParams { width: u32, diff --git a/crates/j2k-jpeg-metal/src/compute/fast_packets/descriptors.rs b/crates/j2k-jpeg-metal/src/compute/fast_packets/descriptors.rs index c5111807d..0496d2753 100644 --- a/crates/j2k-jpeg-metal/src/compute/fast_packets/descriptors.rs +++ b/crates/j2k-jpeg-metal/src/compute/fast_packets/descriptors.rs @@ -6,6 +6,24 @@ use super::super::{ JpegFast422PacketV1, JpegFast444PacketV1, JpegHuffmanTable, MetalRuntime, PixelFormat, PlaneMode, }; +use crate::abi::{CHROMA_FANCY, CHROMA_REPLICATE, CHROMA_UNSAMPLED}; + +/// libjpeg-turbo's chroma geometry for a luma extent at one decode scale: +/// the chroma plane extent the decode kernels write and the pack-kernel chroma +/// mode that maps it onto output pixels. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(in crate::compute) struct ScaledChroma { + pub(in crate::compute) width: u32, + pub(in crate::compute) height: u32, + /// `CHROMA_FANCY`, `CHROMA_REPLICATE` or `CHROMA_UNSAMPLED`. + pub(in crate::compute) mode: u32, +} + +impl ScaledChroma { + pub(in crate::compute) fn plane_len(self) -> usize { + self.width as usize * self.height as usize + } +} /// Chroma geometry for the subsampled families that share the /// `JpegFast420Params` kernel ABI (4:2:0 halves chroma rows, 4:2:2 keeps @@ -19,6 +37,9 @@ pub(in crate::compute) trait FastSubsampledPacket { const MCU_HEIGHT: u32; /// Whether the full-RGB batch path may group restart-interval packets. const FULL_RGB_BATCH_SUPPORTS_RESTART: bool; + /// Whether region windows need a neighbouring pixel on each side for + /// chroma smoothing (the subsampled families). + const REGION_CHROMA_CONTEXT: bool; const ENTROPY_PAYLOAD_CTX: &'static str; const REGION_SCALED_BATCH_OUT_STRIDE_CTX: &'static str; const OUTPUT_STRIDE_CTX: &'static str; @@ -54,6 +75,10 @@ pub(in crate::compute) trait FastSubsampledPacket { fn chroma_width(width: u32) -> u32; fn chroma_height(height: u32) -> u32; + /// Chroma geometry for the luma `extent` of an image `image_width` pixels + /// wide decoded at `scale_shift` (0 full size, 3 = 1/8), following + /// libjpeg-turbo's per-component IDCT size and upsampler choice. + fn scaled_chroma(image_width: u32, extent: (u32, u32), scale_shift: u32) -> ScaledChroma; /// Vertical dispatch extent for the full-frame pack kernels: 4:2:0 packs /// 2x2 pixel quads per thread, 4:2:2 packs 2x1 pairs (full-height rows). fn packed_height_extent(height: u32) -> u32; @@ -134,6 +159,7 @@ macro_rules! impl_fast_subsampled_packet_accessors { impl FastSubsampledPacket for JpegFast420PacketV1 { const FAMILY_NAME: &'static str = "fast420"; + const REGION_CHROMA_CONTEXT: bool = true; const MCU_WIDTH: u32 = 16; const MCU_HEIGHT: u32 = 16; const FULL_RGB_BATCH_SUPPORTS_RESTART: bool = true; @@ -153,6 +179,22 @@ impl FastSubsampledPacket for JpegFast420PacketV1 { fn chroma_height(height: u32) -> u32 { height.div_ceil(2) } + fn scaled_chroma(image_width: u32, extent: (u32, u32), scale_shift: u32) -> ScaledChroma { + if scale_shift > 0 { + // Below full size libjpeg-turbo decodes 4:2:0 chroma with the next + // larger IDCT, at luma resolution. + return ScaledChroma { + width: extent.0, + height: extent.1, + mode: CHROMA_UNSAMPLED, + }; + } + ScaledChroma { + width: Self::chroma_width(extent.0), + height: Self::chroma_height(extent.1), + mode: subsampled_chroma_mode(image_width.div_ceil(2), true), + } + } fn packed_height_extent(height: u32) -> u32 { height.div_ceil(2).max(1) } @@ -160,6 +202,7 @@ impl FastSubsampledPacket for JpegFast420PacketV1 { impl FastSubsampledPacket for JpegFast422PacketV1 { const FAMILY_NAME: &'static str = "fast422"; + const REGION_CHROMA_CONTEXT: bool = true; const MCU_WIDTH: u32 = 16; const MCU_HEIGHT: u32 = 8; const FULL_RGB_BATCH_SUPPORTS_RESTART: bool = false; @@ -179,6 +222,15 @@ impl FastSubsampledPacket for JpegFast422PacketV1 { fn chroma_height(height: u32) -> u32 { height } + fn scaled_chroma(image_width: u32, extent: (u32, u32), scale_shift: u32) -> ScaledChroma { + // `downsampled_width` of 4:2:2 chroma: ceil(width * (8 >> shift) / 16). + let image_chroma_width = image_width.div_ceil(2 << scale_shift); + ScaledChroma { + width: Self::chroma_width(extent.0), + height: Self::chroma_height(extent.1), + mode: subsampled_chroma_mode(image_chroma_width, scale_shift < 3), + } + } fn packed_height_extent(height: u32) -> u32 { height } @@ -186,6 +238,7 @@ impl FastSubsampledPacket for JpegFast422PacketV1 { impl FastSubsampledPacket for JpegFast444PacketV1 { const FAMILY_NAME: &'static str = "fast444"; + const REGION_CHROMA_CONTEXT: bool = false; const MCU_WIDTH: u32 = 8; const MCU_HEIGHT: u32 = 8; const FULL_RGB_BATCH_SUPPORTS_RESTART: bool = false; @@ -205,6 +258,13 @@ impl FastSubsampledPacket for JpegFast444PacketV1 { fn chroma_height(height: u32) -> u32 { height } + fn scaled_chroma(_image_width: u32, extent: (u32, u32), _scale_shift: u32) -> ScaledChroma { + ScaledChroma { + width: extent.0, + height: extent.1, + mode: CHROMA_UNSAMPLED, + } + } fn packed_height_extent(height: u32) -> u32 { height } @@ -213,6 +273,16 @@ impl FastSubsampledPacket for JpegFast444PacketV1 { } } +/// `jdsample.c`: 2:1 chroma is smoothed only above 1/8 scale and when it is +/// more than two samples wide; otherwise each sample is replicated. +fn subsampled_chroma_mode(image_chroma_width: u32, smoothing_scale: bool) -> u32 { + if smoothing_scale && image_chroma_width > 2 { + CHROMA_FANCY + } else { + CHROMA_REPLICATE + } +} + /// Scratch-pool cache keys for one batch driver's buffers; keys stay /// per-family so pooled buffers are never shared across kernel families. #[cfg(target_os = "macos")] diff --git a/crates/j2k-jpeg-metal/src/compute/fast_packets/params.rs b/crates/j2k-jpeg-metal/src/compute/fast_packets/params.rs index 1e92e56dd..1da68dd30 100644 --- a/crates/j2k-jpeg-metal/src/compute/fast_packets/params.rs +++ b/crates/j2k-jpeg-metal/src/compute/fast_packets/params.rs @@ -6,7 +6,7 @@ use super::super::{ JpegFast420WindowedPackParams, JpegFast444PacketV1, JpegFast444Params, JpegFast444ScaledParams, PixelFormat, }; -use super::descriptors::FastSubsampledPacket; +use super::descriptors::{FastSubsampledPacket, ScaledChroma}; use crate::buffers::{checked_copy_bytes_to_buffer_at, new_shared_buffer}; use j2k_core::accelerator::GpuAbi; @@ -21,11 +21,12 @@ pub(in crate::compute) fn fast_subsampled_params( ), })?; let out_stride = packet.dimensions().0 as usize * fmt.bytes_per_pixel(); + let chroma = P::scaled_chroma(packet.dimensions().0, packet.dimensions(), 0); Ok(JpegFast420Params { width: packet.dimensions().0, height: packet.dimensions().1, - chroma_width: P::chroma_width(packet.dimensions().0), - chroma_height: P::chroma_height(packet.dimensions().1), + chroma_width: chroma.width, + chroma_height: chroma.height, mcus_per_row: packet.mcus_per_row(), mcu_rows: packet.mcu_rows(), restart_interval_mcus: packet.restart_interval_mcus(), @@ -41,6 +42,7 @@ pub(in crate::compute) fn fast_subsampled_params( out_format, origin_x: 0, origin_y: 0, + chroma_mode: chroma.mode, }) } @@ -56,11 +58,12 @@ pub(in crate::compute) fn fast_subsampled_region_params ), })?; let out_stride = source_window.w as usize * fmt.bytes_per_pixel(); + let chroma = P::scaled_chroma(packet.dimensions().0, (source_window.w, source_window.h), 0); Ok(JpegFast420Params { width: source_window.w, height: source_window.h, - chroma_width: P::chroma_width(source_window.w), - chroma_height: P::chroma_height(source_window.h), + chroma_width: chroma.width, + chroma_height: chroma.height, mcus_per_row: packet.mcus_per_row(), mcu_rows: packet.mcu_rows(), restart_interval_mcus: packet.restart_interval_mcus(), @@ -76,6 +79,7 @@ pub(in crate::compute) fn fast_subsampled_region_params out_format, origin_x: source_window.x, origin_y: source_window.y, + chroma_mode: chroma.mode, }) } @@ -93,11 +97,16 @@ pub(in crate::compute) fn fast_subsampled_scaled_params let denom = 1u32 << scale_shift; let scaled_width = packet.dimensions().0.div_ceil(denom); let scaled_height = packet.dimensions().1.div_ceil(denom); + let chroma = P::scaled_chroma( + packet.dimensions().0, + (scaled_width, scaled_height), + scale_shift, + ); Some(JpegFast420ScaledParams { scaled_width, scaled_height, - chroma_width: P::chroma_width(scaled_width), - chroma_height: P::chroma_height(scaled_height), + chroma_width: chroma.width, + chroma_height: chroma.height, mcus_per_row: packet.mcus_per_row(), mcu_rows: packet.mcu_rows(), restart_interval_mcus: packet.restart_interval_mcus(), @@ -121,21 +130,49 @@ pub(in crate::compute) fn fast_subsampled_scaled_region_params Option { let full = fast_subsampled_scaled_params(packet, scale)?; + let chroma = P::scaled_chroma( + packet.dimensions().0, + (source_window.w, source_window.h), + full.scale_shift, + ); Some(JpegFast420ScaledParams { scaled_width: source_window.w, scaled_height: source_window.h, - chroma_width: P::chroma_width(source_window.w), - chroma_height: P::chroma_height(source_window.h), + chroma_width: chroma.width, + chroma_height: chroma.height, origin_x: source_window.x, origin_y: source_window.y, ..full }) } +/// `roi` grown by the one pixel of neighbouring chroma the 2:1 smoothing +/// filters read on each side, so a window edge inside the image never stands +/// in for the image edge. 4:4:4 needs no context. +fn roi_with_chroma_context( + dims: (u32, u32), + roi: j2k_jpeg::Rect, +) -> j2k_jpeg::Rect { + if !P::REGION_CHROMA_CONTEXT { + return roi; + } + let x0 = roi.x.saturating_sub(1); + let y0 = roi.y.saturating_sub(1); + let x1 = roi.x.saturating_add(roi.w).saturating_add(1).min(dims.0); + let y1 = roi.y.saturating_add(roi.h).saturating_add(1).min(dims.1); + j2k_jpeg::Rect { + x: x0, + y: y0, + w: x1.saturating_sub(x0), + h: y1.saturating_sub(y0), + } +} + pub(in crate::compute) fn fast_subsampled_full_mcu_window( dims: (u32, u32), roi: j2k_jpeg::Rect, ) -> j2k_jpeg::Rect { + let roi = roi_with_chroma_context::

(dims, roi); let x0 = (roi.x / P::MCU_WIDTH) * P::MCU_WIDTH; let y0 = (roi.y / P::MCU_HEIGHT) * P::MCU_HEIGHT; let x1 = (roi.x + roi.w).div_ceil(P::MCU_WIDTH) * P::MCU_WIDTH; @@ -153,6 +190,7 @@ pub(in crate::compute) fn fast_subsampled_full_mcu_scaled_window j2k_jpeg::Rect { + let roi = roi_with_chroma_context::

(scaled_dims, roi); let mcu_width = P::MCU_WIDTH >> scale_shift; let mcu_height = P::MCU_HEIGHT >> scale_shift; let x0 = (roi.x / mcu_width) * mcu_width; @@ -252,8 +290,11 @@ pub(in crate::compute) fn fast444_scaled_region_params( } #[cfg(target_os = "macos")] +/// Pack parameters for the `roi` of a decoded `dims` window whose chroma +/// planes have the geometry `chroma`. pub(in crate::compute) fn fast_subsampled_windowed_pack_params_for_dims( dims: (u32, u32), + chroma: ScaledChroma, fmt: PixelFormat, roi: j2k_jpeg::Rect, ) -> Result { @@ -267,8 +308,8 @@ pub(in crate::compute) fn fast_subsampled_windowed_pack_params_for_dims( (source_window.w, source_window.h), + chroma, fmt, local_roi, )?; let y_len = source_window.w as usize * source_window.h as usize; - let chroma_len = - source_window.w.div_ceil(2) as usize * P::chroma_height(source_window.h) as usize; + let chroma_len = chroma.plane_len(); let y_plane = if let Some(scratch) = scratch.as_deref_mut() { scratch.private_buffer(&runtime.device, "single_decode_y", y_len)? } else { @@ -291,6 +292,11 @@ pub(in crate::compute) fn encode_fast_subsampled_scaled_batch_item( (source_window.w, source_window.h), + chroma, fmt, local_roi, )?; let y_len = source_window.w as usize * source_window.h as usize; - let chroma_len = - source_window.w.div_ceil(2) as usize * P::chroma_height(source_window.h) as usize; + let chroma_len = chroma.plane_len(); let y_plane = new_decode_plane_buffer(&runtime.device, y_len, false)?; let cb_plane = new_private_buffer(&runtime.device, chroma_len)?; let cr_plane = new_private_buffer(&runtime.device, chroma_len)?; diff --git a/crates/j2k-jpeg-metal/src/compute/region_scaled_plan.rs b/crates/j2k-jpeg-metal/src/compute/region_scaled_plan.rs index 3081a12a3..b853ed2bf 100644 --- a/crates/j2k-jpeg-metal/src/compute/region_scaled_plan.rs +++ b/crates/j2k-jpeg-metal/src/compute/region_scaled_plan.rs @@ -37,6 +37,7 @@ pub(super) fn windowed_texture_pack_params( height: plan.pack_params.height, tile_index: 0, alpha: u32::from(u8::MAX), + chroma_mode: plan.pack_params.chroma_mode, } } @@ -174,8 +175,14 @@ pub(super) fn fast_subsampled_region_scaled_batch_plan( w: scaled_roi.w, h: scaled_roi.h, }; + let chroma = P::scaled_chroma( + packet.dimensions().0, + (source_window.w, source_window.h), + full_params.scale_shift, + ); let pack_params = fast_subsampled_windowed_pack_params_for_dims::

( (source_window.w, source_window.h), + chroma, PixelFormat::Rgb8, local_roi, ) @@ -209,10 +216,10 @@ pub(super) fn fast_subsampled_region_scaled_batch_plan( alpha: u32::from(u8::MAX), mode: plane_mode_to_u32(mode), out_format: OUT_RGB, + chroma_mode: pack_params.chroma_mode, }, y_len: source_window.w as usize * source_window.h as usize, - chroma_len: P::chroma_width(source_window.w) as usize - * P::chroma_height(source_window.h) as usize, + chroma_len: chroma.plane_len(), out_tile_len: out_stride * scaled_roi.h as usize, out_dims: (scaled_roi.w, scaled_roi.h), }) diff --git a/crates/j2k-jpeg-metal/src/compute/single_decode/subsampled.rs b/crates/j2k-jpeg-metal/src/compute/single_decode/subsampled.rs index c716a0f17..40136e31f 100644 --- a/crates/j2k-jpeg-metal/src/compute/single_decode/subsampled.rs +++ b/crates/j2k-jpeg-metal/src/compute/single_decode/subsampled.rs @@ -399,14 +399,19 @@ fn try_decode_fast_subsampled_scaled_region_to_surface( w: scaled_roi.w, h: scaled_roi.h, }; + let chroma = P::scaled_chroma( + packet.dimensions().0, + (source_window.w, source_window.h), + full_params.scale_shift, + ); let pack_params = fast_subsampled_windowed_pack_params_for_dims::

( (source_window.w, source_window.h), + chroma, fmt, local_roi, )?; let y_len = source_window.w as usize * source_window.h as usize; - let chroma_len = - source_window.w.div_ceil(2) as usize * P::chroma_height(source_window.h) as usize; + let chroma_len = chroma.plane_len(); let mut scratch = runtime.batch_scratch()?; let y_plane = scratch.private_buffer(&runtime.device, "single_decode_y", y_len)?; let cb_plane = scratch.private_buffer(&runtime.device, "single_decode_cb", chroma_len)?; diff --git a/crates/j2k-jpeg-metal/src/shaders_decode_fast422_regions.metal b/crates/j2k-jpeg-metal/src/shaders_decode_fast422_regions.metal index f529fcb82..549627cc0 100644 --- a/crates/j2k-jpeg-metal/src/shaders_decode_fast422_regions.metal +++ b/crates/j2k-jpeg-metal/src/shaders_decode_fast422_regions.metal @@ -743,8 +743,12 @@ kernel void jpeg_decode_fast420_scaled( thread short coeffs[64]; const uint y_block_size = 8u >> params.scale_shift; - const uint c_block_size = 8u >> params.scale_shift; const uint y_mcu_size = 16u >> params.scale_shift; + // Below full size libjpeg-turbo decodes 4:2:0 chroma with the next larger + // reduced IDCT, so each chroma block covers its whole MCU and the chroma + // planes are at output resolution (the pack kernels use CHROMA_UNSAMPLED). + const uint c_scale_shift = params.scale_shift == 0u ? 0u : params.scale_shift - 1u; + const uint c_block_size = 8u >> c_scale_shift; uint mx = 0u; uint my = 0u; @@ -766,10 +770,10 @@ kernel void jpeg_decode_fast420_scaled( if (!jpeg_decode_deposit_scaled_block(br, entropy, params.entropy_len, y_dc, y_ac, y_quant, y_prev_dc, thread_status, y_plane, params.scaled_width, params.scaled_width, params.scaled_height, y_x + y_block_size, y_y + y_block_size, params.scale_shift, coeffs)) { return; } - if (!jpeg_decode_deposit_scaled_block(br, entropy, params.entropy_len, cb_dc, cb_ac, cb_quant, cb_prev_dc, thread_status, cb_plane, params.chroma_width, params.chroma_width, params.chroma_height, c_x, c_y, params.scale_shift, coeffs)) { + if (!jpeg_decode_deposit_scaled_block(br, entropy, params.entropy_len, cb_dc, cb_ac, cb_quant, cb_prev_dc, thread_status, cb_plane, params.chroma_width, params.chroma_width, params.chroma_height, c_x, c_y, c_scale_shift, coeffs)) { return; } - if (!jpeg_decode_deposit_scaled_block(br, entropy, params.entropy_len, cr_dc, cr_ac, cr_quant, cr_prev_dc, thread_status, cr_plane, params.chroma_width, params.chroma_width, params.chroma_height, c_x, c_y, params.scale_shift, coeffs)) { + if (!jpeg_decode_deposit_scaled_block(br, entropy, params.entropy_len, cr_dc, cr_ac, cr_quant, cr_prev_dc, thread_status, cr_plane, params.chroma_width, params.chroma_width, params.chroma_height, c_x, c_y, c_scale_shift, coeffs)) { return; } advance_mcu_cursor(mx, my, params.mcus_per_row); @@ -814,10 +818,14 @@ kernel void jpeg_decode_fast420_scaled_region( thread short coeffs[64]; const uint y_block_size = 8u >> params.scale_shift; - const uint c_block_size = 8u >> params.scale_shift; const uint y_mcu_size = 16u >> params.scale_shift; - const uint chroma_origin_x = params.origin_x / 2u; - const uint chroma_origin_y = params.origin_y / 2u; + // Below full size libjpeg-turbo decodes 4:2:0 chroma with the next larger + // reduced IDCT: chroma blocks cover whole MCUs at output resolution. + const bool chroma_unsampled = params.scale_shift != 0u; + const uint c_scale_shift = chroma_unsampled ? params.scale_shift - 1u : 0u; + const uint c_block_size = 8u >> c_scale_shift; + const uint chroma_origin_x = chroma_unsampled ? params.origin_x : params.origin_x / 2u; + const uint chroma_origin_y = chroma_unsampled ? params.origin_y : params.origin_y / 2u; uint mx = 0u; uint my = 0u; @@ -839,10 +847,10 @@ kernel void jpeg_decode_fast420_scaled_region( if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, params.entropy_len, y_dc, y_ac, y_quant, y_prev_dc, thread_status, y_plane, params.scaled_width, params.scaled_width, params.scaled_height, params.origin_x, params.origin_y, y_x + y_block_size, y_y + y_block_size, y_block_size, y_block_size, params.scale_shift, coeffs)) { return; } - if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, params.entropy_len, cb_dc, cb_ac, cb_quant, cb_prev_dc, thread_status, cb_plane, params.chroma_width, params.chroma_width, params.chroma_height, chroma_origin_x, chroma_origin_y, c_x, c_y, c_block_size, c_block_size, params.scale_shift, coeffs)) { + if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, params.entropy_len, cb_dc, cb_ac, cb_quant, cb_prev_dc, thread_status, cb_plane, params.chroma_width, params.chroma_width, params.chroma_height, chroma_origin_x, chroma_origin_y, c_x, c_y, c_block_size, c_block_size, c_scale_shift, coeffs)) { return; } - if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, params.entropy_len, cr_dc, cr_ac, cr_quant, cr_prev_dc, thread_status, cr_plane, params.chroma_width, params.chroma_width, params.chroma_height, chroma_origin_x, chroma_origin_y, c_x, c_y, c_block_size, c_block_size, params.scale_shift, coeffs)) { + if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, params.entropy_len, cr_dc, cr_ac, cr_quant, cr_prev_dc, thread_status, cr_plane, params.chroma_width, params.chroma_width, params.chroma_height, chroma_origin_x, chroma_origin_y, c_x, c_y, c_block_size, c_block_size, c_scale_shift, coeffs)) { return; } advance_mcu_cursor(mx, my, params.mcus_per_row); @@ -895,10 +903,14 @@ kernel void jpeg_decode_fast420_scaled_region_batch( thread short coeffs[64]; const uint y_block_size = 8u >> params.scale_shift; - const uint c_block_size = 8u >> params.scale_shift; const uint y_mcu_size = 16u >> params.scale_shift; - const uint chroma_origin_x = params.origin_x / 2u; - const uint chroma_origin_y = params.origin_y / 2u; + // Below full size libjpeg-turbo decodes 4:2:0 chroma with the next larger + // reduced IDCT: chroma blocks cover whole MCUs at output resolution. + const bool chroma_unsampled = params.scale_shift != 0u; + const uint c_scale_shift = chroma_unsampled ? params.scale_shift - 1u : 0u; + const uint c_block_size = 8u >> c_scale_shift; + const uint chroma_origin_x = chroma_unsampled ? params.origin_x : params.origin_x / 2u; + const uint chroma_origin_y = chroma_unsampled ? params.origin_y : params.origin_y / 2u; uint mx = 0u; uint my = 0u; @@ -920,10 +932,10 @@ kernel void jpeg_decode_fast420_scaled_region_batch( if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, entropy_end, y_dc, y_ac, y_quant, y_prev_dc, thread_status, tile_y_plane, params.scaled_width, params.scaled_width, params.scaled_height, params.origin_x, params.origin_y, y_x + y_block_size, y_y + y_block_size, y_block_size, y_block_size, params.scale_shift, coeffs)) { return; } - if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, entropy_end, cb_dc, cb_ac, cb_quant, cb_prev_dc, thread_status, tile_cb_plane, params.chroma_width, params.chroma_width, params.chroma_height, chroma_origin_x, chroma_origin_y, c_x, c_y, c_block_size, c_block_size, params.scale_shift, coeffs)) { + if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, entropy_end, cb_dc, cb_ac, cb_quant, cb_prev_dc, thread_status, tile_cb_plane, params.chroma_width, params.chroma_width, params.chroma_height, chroma_origin_x, chroma_origin_y, c_x, c_y, c_block_size, c_block_size, c_scale_shift, coeffs)) { return; } - if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, entropy_end, cr_dc, cr_ac, cr_quant, cr_prev_dc, thread_status, tile_cr_plane, params.chroma_width, params.chroma_width, params.chroma_height, chroma_origin_x, chroma_origin_y, c_x, c_y, c_block_size, c_block_size, params.scale_shift, coeffs)) { + if (!jpeg_decode_deposit_scaled_region_block_or_skip(br, entropy, entropy_end, cr_dc, cr_ac, cr_quant, cr_prev_dc, thread_status, tile_cr_plane, params.chroma_width, params.chroma_width, params.chroma_height, chroma_origin_x, chroma_origin_y, c_x, c_y, c_block_size, c_block_size, c_scale_shift, coeffs)) { return; } advance_mcu_cursor(mx, my, params.mcus_per_row); diff --git a/crates/j2k-jpeg-metal/src/shaders_encode.metal b/crates/j2k-jpeg-metal/src/shaders_encode.metal index 807c8b650..d5614edca 100644 --- a/crates/j2k-jpeg-metal/src/shaders_encode.metal +++ b/crates/j2k-jpeg-metal/src/shaders_encode.metal @@ -586,15 +586,24 @@ inline bool block_intersects_rect( return block_x < rect_x1 && rect_x < block_x1 && block_y < rect_y1 && rect_y < block_y1; } -inline int descale(int value, int shift) { - return value >> shift; -} - +// Callers of descale_and_clamp add their own rounding term (the full islow IDCT +// folds it into the DC path). inline uchar descale_and_clamp(int value, int shift) { const int shifted = value >> shift; return clamp_u8(shifted + 128); } +// libjpeg DESCALE: arithmetic right shift that rounds half up. Every stage of +// jidctred.c (and libjpeg-turbo's SIMD ports) rounds; a plain shift biases +// DCT-scaled output low by up to one level per component. +inline int descale_rounded(int value, int shift) { + return (value + (1 << (shift - 1))) >> shift; +} + +inline uchar descale_rounded_and_clamp(int value, int shift) { + return clamp_u8(descale_rounded(value, shift) + 128); +} + // libjpeg islow column pass over four columns at once. There is no per-column // DC shortcut: with all AC terms zero the full expression is exactly p0 << PASS1_BITS, // and i16 inputs cannot overflow it. @@ -951,10 +960,10 @@ inline void idct_4x4_column( + p1 * FIX_2_562915447; const int shift = CONST_BITS - PASS1_BITS + 1; - work[col] = descale(tmp10 + tmp2, shift); - work[24 + col] = descale(tmp10 - tmp2, shift); - work[8 + col] = descale(tmp12 + tmp0, shift); - work[16 + col] = descale(tmp12 - tmp0, shift); + work[col] = descale_rounded(tmp10 + tmp2, shift); + work[24 + col] = descale_rounded(tmp10 - tmp2, shift); + work[8 + col] = descale_rounded(tmp12 + tmp0, shift); + work[16 + col] = descale_rounded(tmp12 - tmp0, shift); } inline void idct_4x4_row( @@ -973,7 +982,7 @@ inline void idct_4x4_row( const uint out = row * 4; if (p1 == 0 && p2 == 0 && p3 == 0 && p5 == 0 && p6 == 0 && p7 == 0) { - const uchar dc = descale_and_clamp(p0, PASS1_BITS + 3); + const uchar dc = descale_rounded_and_clamp(p0, PASS1_BITS + 3); output[out] = dc; output[out + 1] = dc; output[out + 2] = dc; @@ -996,10 +1005,10 @@ inline void idct_4x4_row( + p1 * FIX_2_562915447; const int shift = CONST_BITS + PASS1_BITS + 3 + 1; - output[out] = descale_and_clamp(tmp10 + tmp2, shift); - output[out + 3] = descale_and_clamp(tmp10 - tmp2, shift); - output[out + 1] = descale_and_clamp(tmp12 + tmp0, shift); - output[out + 2] = descale_and_clamp(tmp12 - tmp0, shift); + output[out] = descale_rounded_and_clamp(tmp10 + tmp2, shift); + output[out + 3] = descale_rounded_and_clamp(tmp10 - tmp2, shift); + output[out + 1] = descale_rounded_and_clamp(tmp12 + tmp0, shift); + output[out + 2] = descale_rounded_and_clamp(tmp12 - tmp0, shift); } inline void idct_islow_4x4( @@ -1043,8 +1052,8 @@ inline void idct_2x2_column( + p1 * FIX_3_624509785; const int shift = CONST_BITS - PASS1_BITS + 2; - work[col] = descale(tmp10 + tmp0, shift); - work[8 + col] = descale(tmp10 - tmp0, shift); + work[col] = descale_rounded(tmp10 + tmp0, shift); + work[8 + col] = descale_rounded(tmp10 - tmp0, shift); } inline void idct_2x2_row( @@ -1060,7 +1069,7 @@ inline void idct_2x2_row( const int p7 = work[base + 7]; if (p1 == 0 && p3 == 0 && p5 == 0 && p7 == 0) { - const uchar dc = descale_and_clamp(p0, PASS1_BITS + 3); + const uchar dc = descale_rounded_and_clamp(p0, PASS1_BITS + 3); const uint out = row * 2; output[out] = dc; output[out + 1] = dc; @@ -1075,8 +1084,8 @@ inline void idct_2x2_row( const int shift = CONST_BITS + PASS1_BITS + 5; const uint out = row * 2; - output[out] = descale_and_clamp(tmp10 + tmp0, shift); - output[out + 1] = descale_and_clamp(tmp10 - tmp0, shift); + output[out] = descale_rounded_and_clamp(tmp10 + tmp0, shift); + output[out + 1] = descale_rounded_and_clamp(tmp10 - tmp0, shift); } inline void idct_islow_2x2( @@ -1096,7 +1105,7 @@ inline void idct_islow_2x2( } inline uchar idct_islow_1x1(thread const short input[64]) { - return descale_and_clamp(int(input[0]), 3); + return descale_rounded_and_clamp(int(input[0]), 3); } inline void deposit_block_region( @@ -1238,6 +1247,13 @@ inline void deposit_scaled_block( thread const short coeffs[64], bool dc_only ) { + if (scale_shift == 0u) { + // 4:2:0 chroma at 1/2 scale: libjpeg-turbo's full 8x8 IDCT. + thread uchar pixels8[64]; + idct_block(coeffs, dc_only, pixels8); + deposit_block_region(plane, stride, width, height, 0u, 0u, x, y, pixels8); + return; + } if (scale_shift == 1u) { thread uchar pixels4[16]; if (dc_only) { @@ -1643,12 +1659,25 @@ inline void jpeg_sample_420_chroma( device const uchar *cr_plane, uint chroma_width, uint chroma_height, + uint chroma_mode, uint x, uint y, thread uchar &cb, thread uchar &cr ) { + if (chroma_mode == CHROMA_UNSAMPLED) { + const uint idx = min(y, chroma_height - 1u) * chroma_width + min(x, chroma_width - 1u); + cb = cb_plane[idx]; + cr = cr_plane[idx]; + return; + } const uint chroma_y = min(y / 2u, chroma_height - 1u); + if (chroma_mode == CHROMA_REPLICATE) { + const uint idx = chroma_y * chroma_width + min(x / 2u, chroma_width - 1u); + cb = cb_plane[idx]; + cr = cr_plane[idx]; + return; + } const uint near_y = (y & 1u) == 0u ? (chroma_y == 0u ? 0u : chroma_y - 1u) : min(chroma_y + 1u, chroma_height - 1u); @@ -1666,6 +1695,7 @@ inline void jpeg_sample_422_chroma( device const uchar *cr_plane, uint chroma_width, uint chroma_height, + uint chroma_mode, uint x, uint y, thread uchar &cb, @@ -1674,6 +1704,12 @@ inline void jpeg_sample_422_chroma( const uint chroma_y = min(y, chroma_height - 1u); device const uchar *curr_cb = cb_plane + chroma_y * chroma_width; device const uchar *curr_cr = cr_plane + chroma_y * chroma_width; + if (chroma_mode != CHROMA_FANCY) { + const uint chroma_x = min(chroma_mode == CHROMA_UNSAMPLED ? x : x / 2u, chroma_width - 1u); + cb = curr_cb[chroma_x]; + cr = curr_cr[chroma_x]; + return; + } cb = h2v1_sample(curr_cb, chroma_width, x); cr = h2v1_sample(curr_cr, chroma_width, x); diff --git a/crates/j2k-jpeg-metal/src/shaders_pack_subsampled.metal b/crates/j2k-jpeg-metal/src/shaders_pack_subsampled.metal index 65454505d..1d08e9c02 100644 --- a/crates/j2k-jpeg-metal/src/shaders_pack_subsampled.metal +++ b/crates/j2k-jpeg-metal/src/shaders_pack_subsampled.metal @@ -23,6 +23,8 @@ kernel void jpeg_pack_420( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, gid.x, gid.y, cb, @@ -56,6 +58,8 @@ kernel void jpeg_pack_420_rgb( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, gid.x, gid.y, cb, @@ -157,6 +161,8 @@ kernel void jpeg_pack_420_rgba( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, gid.x, gid.y, cb, @@ -266,6 +272,8 @@ kernel void jpeg_pack_422_rgb( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, gid.x, gid.y, cb, @@ -380,6 +388,8 @@ kernel void jpeg_pack_422_rgba( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, gid.x, gid.y, cb, @@ -421,6 +431,8 @@ kernel void jpeg_pack_422_windowed( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, @@ -460,6 +472,8 @@ kernel void jpeg_pack_422_windowed_rgb( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, @@ -502,6 +516,8 @@ kernel void jpeg_pack_422_windowed_rgb_batch( tile_cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, @@ -545,6 +561,8 @@ kernel void jpeg_pack_422_windowed_rgba_texture( tile_cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, @@ -579,6 +597,8 @@ kernel void jpeg_pack_422_windowed_rgba( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, @@ -620,6 +640,8 @@ kernel void jpeg_pack_420_windowed( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, @@ -659,6 +681,8 @@ kernel void jpeg_pack_420_windowed_rgb( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, @@ -701,6 +725,8 @@ kernel void jpeg_pack_420_windowed_rgb_batch( tile_cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, @@ -744,6 +770,8 @@ kernel void jpeg_pack_420_windowed_rgba_texture( tile_cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, @@ -797,6 +825,8 @@ kernel void jpeg_pack_420_windowed_rgba( cr_plane, params.chroma_width, params.chroma_height, + + params.chroma_mode, src_x, src_y, cb, diff --git a/crates/j2k-jpeg-metal/src/shaders_shared.metal b/crates/j2k-jpeg-metal/src/shaders_shared.metal index 25b230dd6..c237a1499 100644 --- a/crates/j2k-jpeg-metal/src/shaders_shared.metal +++ b/crates/j2k-jpeg-metal/src/shaders_shared.metal @@ -71,6 +71,7 @@ struct JpegFast420Params { uint out_format; uint origin_x; uint origin_y; + uint chroma_mode; }; struct JpegFast420ScaledParams { @@ -128,6 +129,7 @@ struct JpegFast420WindowedPackParams { uint out_stride; uint alpha; uint out_format; + uint chroma_mode; }; struct JpegFast420BatchParams { @@ -206,6 +208,7 @@ struct JpegWindowedPackBatchParams { uint alpha; uint mode; uint out_format; + uint chroma_mode; }; struct JpegWindowedTexturePackBatchParams { @@ -219,6 +222,7 @@ struct JpegWindowedTexturePackBatchParams { uint height; uint tile_index; uint alpha; + uint chroma_mode; }; struct JpegTexturePackBatchParams { @@ -280,6 +284,17 @@ constant uint MODE_GRAY = 0; constant uint MODE_YCBCR = 1; constant uint MODE_RGB = 2; +// How the pack kernels map subsampled chroma planes onto output pixels, +// following libjpeg-turbo's upsampler choice for the decode scale. +// CHROMA_FANCY: triangle-filter upsampling (h2v1/h2v2_fancy_upsample). +// CHROMA_REPLICATE: each chroma sample repeated (smoothing off at 1/8 scale or +// for chroma at most two samples wide). +// CHROMA_UNSAMPLED: chroma planes already at output resolution (4:2:0 decoded +// below full size with libjpeg-turbo's larger chroma IDCT). +constant uint CHROMA_FANCY = 0; +constant uint CHROMA_REPLICATE = 1; +constant uint CHROMA_UNSAMPLED = 2; + constant uint OUT_GRAY = 0; constant uint OUT_RGB = 1; constant uint OUT_RGBA = 2; diff --git a/crates/j2k-jpeg-metal/tests/scaled_matrix.rs b/crates/j2k-jpeg-metal/tests/scaled_matrix.rs new file mode 100644 index 000000000..82590f7be --- /dev/null +++ b/crates/j2k-jpeg-metal/tests/scaled_matrix.rs @@ -0,0 +1,321 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 + +//! Metal DCT-scaled decodes against the libjpeg-turbo reference matrix. +//! +//! The Metal fast families (8-bit baseline 4:4:4, 4:2:2 and 4:2:0) must +//! reproduce libjpeg-turbo's scaled output exactly: 4:2:0 chroma decoded with +//! the larger IDCT below full size, 4:2:2 chroma replicated at 1/8 and when at +//! most two samples wide. Images at most four pixels wide, whose full-size +//! 2:1 chroma libjpeg-turbo replicates, are not Metal fast shapes and must be +//! rejected by explicit Metal requests rather than decoded differently. +#![cfg(target_os = "macos")] + +use j2k_core::{BackendKind, BackendRequest, DeviceSurface, Downscale, PixelFormat, Rect}; +use j2k_jpeg_metal::{Decoder, JpegTileBatch, MetalDecodeRequest}; +use j2k_test_support::{ + crop_interleaved_bytes, scaled_matrix_cases, scaled_rect_covering, PixelRect, ScaledMatrixCase, + SCALED_MATRIX_DENOMINATORS, +}; + +fn should_run_metal_runtime() -> bool { + j2k_test_support::metal_runtime_gate(module_path!()) +} + +fn downscale(denominator: u32) -> Downscale { + match denominator { + 1 => Downscale::None, + 2 => Downscale::Half, + 4 => Downscale::Quarter, + 8 => Downscale::Eighth, + _ => unreachable!("matrix denominators are 1, 2, 4, 8"), + } +} + +/// 8-bit sequential cases in the Metal fast families. +fn metal_family_cases() -> Vec { + scaled_matrix_cases() + .into_iter() + .filter(|case| { + case.precision == 8 + && matches!(case.layout, "444" | "422" | "420") + && matches!(case.coding, "baseline" | "restart") + }) + .collect() +} + +/// Full-size 2:1 chroma at most two samples wide: libjpeg-turbo replicates +/// it, so these images are not Metal fast shapes. +fn is_narrow_subsampled(case: &ScaledMatrixCase) -> bool { + case.layout != "444" && case.width <= 4 +} + +fn expected_rgba(rgb: &[u8]) -> Vec { + rgb.chunks_exact(3) + .flat_map(|px| [px[0], px[1], px[2], u8::MAX]) + .collect() +} + +fn mismatch(expected: &[u8], actual: &[u8]) -> Option { + if expected.len() != actual.len() { + return Some(format!( + "{} bytes, expected {}", + actual.len(), + expected.len() + )); + } + let diffs: Vec = expected + .iter() + .zip(actual) + .map(|(&e, &a)| e.abs_diff(a)) + .filter(|&d| d > 0) + .collect(); + let max = diffs.iter().copied().max()?; + Some(format!( + "{} of {} bytes differ, max {max}", + diffs.len(), + expected.len() + )) +} + +fn assert_no_failures(failures: &[String]) { + assert!( + failures.is_empty(), + "{} Metal scaled decodes differ from libjpeg-turbo:\n{}", + failures.len(), + failures.join("\n") + ); +} + +fn region_for(case: &ScaledMatrixCase) -> Rect { + Rect { + x: case.width / 3, + y: case.height / 4, + w: case.width / 2, + h: case.height / 2, + } +} + +fn expected_region(case: &ScaledMatrixCase, roi: Rect, denominator: u32) -> (PixelRect, Vec) { + let scaled = scaled_rect_covering( + PixelRect { + x: roi.x, + y: roi.y, + w: roi.w, + h: roi.h, + }, + denominator, + ); + let (width, _) = case.scaled_dimensions(denominator); + let crop = crop_interleaved_bytes(case.reference(denominator), width as usize, 3, scaled); + (scaled, crop) +} + +#[test] +fn metal_scaled_decodes_match_libjpeg_turbo() { + if !should_run_metal_runtime() { + return; + } + let mut failures = Vec::new(); + for case in metal_family_cases() + .iter() + .filter(|case| !is_narrow_subsampled(case)) + { + for denominator in SCALED_MATRIX_DENOMINATORS { + for fmt in [PixelFormat::Rgb8, PixelFormat::Rgba8] { + let label = format!("{} {fmt:?} 1/{denominator}", case.name); + let mut decoder = Decoder::new(case.jpeg).expect("matrix JPEG parses"); + let request = + MetalDecodeRequest::scaled(fmt, downscale(denominator), BackendRequest::Metal); + match decoder.decode_request_to_device(request) { + Ok(surface) => { + assert_eq!(surface.backend_kind(), BackendKind::Metal, "{label}"); + let rgb = case.reference(denominator); + let expected = if fmt == PixelFormat::Rgba8 { + expected_rgba(rgb) + } else { + rgb.to_vec() + }; + let bytes = surface.as_bytes().expect("surface bytes"); + if let Some(message) = mismatch(&expected, &bytes) { + failures.push(format!("{label}: {message}")); + } + } + Err(err) => failures.push(format!("{label}: Metal decode failed: {err}")), + } + } + } + } + assert_no_failures(&failures); +} + +#[test] +fn metal_scaled_region_decodes_match_libjpeg_turbo_crops() { + if !should_run_metal_runtime() { + return; + } + let mut failures = Vec::new(); + for case in metal_family_cases().iter().filter(|case| case.width >= 18) { + let roi = region_for(case); + for denominator in SCALED_MATRIX_DENOMINATORS { + let label = format!("{} region 1/{denominator}", case.name); + let mut decoder = Decoder::new(case.jpeg).expect("matrix JPEG parses"); + let request = MetalDecodeRequest::region_scaled( + PixelFormat::Rgb8, + roi, + downscale(denominator), + BackendRequest::Metal, + ); + let (scaled, expected) = expected_region(case, roi, denominator); + match decoder.decode_request_to_device(request) { + Ok(surface) => { + assert_eq!(surface.dimensions(), (scaled.w, scaled.h), "{label}"); + let bytes = surface.as_bytes().expect("surface bytes"); + if let Some(message) = mismatch(&expected, &bytes) { + failures.push(format!("{label}: {message}")); + } + } + Err(err) => failures.push(format!("{label}: Metal decode failed: {err}")), + } + } + } + assert_no_failures(&failures); +} + +/// Batches group same-shape tiles into shared dispatches; every tile in a +/// batch must still match its own reference. +#[test] +fn metal_scaled_tile_batches_match_libjpeg_turbo() { + if !should_run_metal_runtime() { + return; + } + let cases: Vec = metal_family_cases() + .into_iter() + .filter(|case| !is_narrow_subsampled(case)) + .collect(); + let mut failures = Vec::new(); + for denominator in SCALED_MATRIX_DENOMINATORS { + for region in [false, true] { + let mut batch = JpegTileBatch::with_capacity(cases.len() * 2); + let mut expected = Vec::new(); + for case in cases.iter().filter(|case| !region || case.width >= 18) { + // Two copies so compatible tiles share a grouped dispatch. + for _ in 0..2 { + let request = if region { + let roi = region_for(case); + expected.push((case.name, expected_region(case, roi, denominator).1)); + MetalDecodeRequest::region_scaled( + PixelFormat::Rgb8, + roi, + downscale(denominator), + BackendRequest::Metal, + ) + } else { + expected.push((case.name, case.reference(denominator).to_vec())); + MetalDecodeRequest::scaled( + PixelFormat::Rgb8, + downscale(denominator), + BackendRequest::Metal, + ) + }; + batch + .push_tile_request(case.jpeg, request) + .expect("push batch tile"); + } + } + let label = if region { "region batch" } else { "batch" }; + match batch.decode_all() { + Ok(surfaces) => { + for (surface, (name, expected)) in surfaces.iter().zip(&expected) { + let bytes = surface.as_bytes().expect("surface bytes"); + if let Some(message) = mismatch(expected, &bytes) { + failures.push(format!("{name} {label} 1/{denominator}: {message}")); + } + } + } + Err(err) => failures.push(format!("{label} 1/{denominator} failed: {err}")), + } + } + } + assert_no_failures(&failures); +} + +#[test] +fn explicit_metal_rejects_narrow_subsampled_images() { + if !should_run_metal_runtime() { + return; + } + for case in metal_family_cases() + .iter() + .filter(|case| is_narrow_subsampled(case)) + { + for denominator in SCALED_MATRIX_DENOMINATORS { + let mut decoder = Decoder::new(case.jpeg).expect("matrix JPEG parses"); + let request = MetalDecodeRequest::scaled( + PixelFormat::Rgb8, + downscale(denominator), + BackendRequest::Metal, + ); + assert!( + decoder.decode_request_to_device(request).is_err(), + "{} 1/{denominator}: narrow 2:1 chroma must not take the Metal fast path", + case.name + ); + } + } +} + +/// Region edges on and beside MCU boundaries at every scale, where the +/// smoothing filters need chroma from outside the decoded window. +#[test] +fn metal_scaled_region_edge_sweep_matches_libjpeg_turbo_crops() { + if !should_run_metal_runtime() { + return; + } + let xs = [0u32, 7, 8, 9, 15, 16, 17]; + let ys = [0u32, 7, 8, 9, 16, 17]; + let mut failures = Vec::new(); + for case in metal_family_cases().iter().filter(|case| case.width == 45) { + for denominator in SCALED_MATRIX_DENOMINATORS { + for &x0 in &xs { + for &x1 in &xs { + for &y0 in &ys { + for &y1 in &ys { + let (x1, y1) = (x1 + 8, y1 + 8); + if x0 >= x1 || y0 >= y1 || x1 > case.width || y1 > case.height { + continue; + } + let roi = Rect { + x: x0, + y: y0, + w: x1 - x0, + h: y1 - y0, + }; + let label = format!( + "{} roi {},{} {}x{} 1/{denominator}", + case.name, roi.x, roi.y, roi.w, roi.h + ); + let mut decoder = Decoder::new(case.jpeg).expect("matrix JPEG parses"); + let request = MetalDecodeRequest::region_scaled( + PixelFormat::Rgb8, + roi, + downscale(denominator), + BackendRequest::Metal, + ); + let (_, expected) = expected_region(case, roi, denominator); + match decoder.decode_request_to_device(request) { + Ok(surface) => { + let bytes = surface.as_bytes().expect("surface bytes"); + if let Some(message) = mismatch(&expected, &bytes) { + failures.push(format!("{label}: {message}")); + } + } + Err(err) => failures.push(format!("{label}: {err}")), + } + } + } + } + } + } + } + assert_no_failures(&failures); +} diff --git a/crates/j2k-jpeg/src/adapter/fast_packet/family.rs b/crates/j2k-jpeg/src/adapter/fast_packet/family.rs index 7e99c4fe4..a5c606c31 100644 --- a/crates/j2k-jpeg/src/adapter/fast_packet/family.rs +++ b/crates/j2k-jpeg/src/adapter/fast_packet/family.rs @@ -29,9 +29,17 @@ pub(super) fn classify_color_fast_packet_info(info: &Info) -> Option 4; match (info.color_space, info.sampling.components()) { - (ColorSpace::YCbCr, [(2, 2), (1, 1), (1, 1)]) => Some(JpegFastPacketFamily::Fast420), - (ColorSpace::YCbCr, [(2, 1), (1, 1), (1, 1)]) => Some(JpegFastPacketFamily::Fast422), + (ColorSpace::YCbCr, [(2, 2), (1, 1), (1, 1)]) if smoothed_chroma => { + Some(JpegFastPacketFamily::Fast420) + } + (ColorSpace::YCbCr, [(2, 1), (1, 1), (1, 1)]) if smoothed_chroma => { + Some(JpegFastPacketFamily::Fast422) + } (ColorSpace::YCbCr | ColorSpace::Rgb, [(1, 1), (1, 1), (1, 1)]) => { Some(JpegFastPacketFamily::Fast444) } diff --git a/crates/j2k-jpeg/src/color/mod.rs b/crates/j2k-jpeg/src/color/mod.rs index b678f469d..9b8f0b957 100644 --- a/crates/j2k-jpeg/src/color/mod.rs +++ b/crates/j2k-jpeg/src/color/mod.rs @@ -3,5 +3,6 @@ //! Color conversion + chroma upsampling. Scalar-only in M1b. pub(crate) mod cmyk; +pub(crate) mod scaled_sampling; pub(crate) mod upsample; pub(crate) mod ycbcr; diff --git a/crates/j2k-jpeg/src/color/scaled_sampling.rs b/crates/j2k-jpeg/src/color/scaled_sampling.rs new file mode 100644 index 000000000..e17252c2f --- /dev/null +++ b/crates/j2k-jpeg/src/color/scaled_sampling.rs @@ -0,0 +1,226 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 + +//! libjpeg-turbo's per-component DCT scaling and upsampler choice. +//! +//! When decoding at a reduced scale, libjpeg-turbo (`jdmaster.c`) gives each +//! component its own reduced IDCT: a subsampled component is decoded with a +//! larger IDCT whenever that replaces upsampling, so 4:2:0 chroma at 1/2 is +//! decoded with a full 8x8 IDCT and needs no upsampling at all. `jdsample.c` +//! then picks each component's upsampler: the smoothing ("fancy") filters +//! apply only above 1/8 scale, and the 2:1 horizontal filters only when the +//! component is more than two samples wide; everything else replicates. +//! +//! Every CPU decode path takes its geometry from [`ScaledSampling`] so all of +//! them reproduce libjpeg-turbo (and therefore `OpenSlide`) bit for bit. + +use crate::info::{DownscaleFactor, SamplingFactors}; + +/// How one component's decoded samples reach the output grid. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) enum Upsample { + /// The component is already at output resolution. + None, + /// `h2v1_fancy_upsample`: 2:1 horizontal triangle filter. + FancyH2V1, + /// `h1v2_fancy_upsample`: 1:2 vertical triangle filter. + FancyH1V2, + /// `h2v2_fancy_upsample`: 2:1 horizontal and vertical triangle filter. + FancyH2V2, + /// `int_upsample` and friends: each sample is repeated `h_ratio` times + /// across and each row `v_ratio` times down. + Replicate, +} + +/// Decode geometry of one component at a DCT scale. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub(crate) struct ScaledComponent { + /// Reduced IDCT output size per block side: 8, 4, 2 or 1. + pub(crate) idct_size: u32, + /// Output pixels per decoded sample, horizontally. + pub(crate) h_ratio: u32, + /// Output pixels per decoded sample, vertically. + pub(crate) v_ratio: u32, + /// Decoded samples across the whole image (`downsampled_width`). + pub(crate) width: u32, + /// Decoded sample rows in the whole image (`downsampled_height`). + pub(crate) height: u32, + /// Upsampler libjpeg-turbo selects for this component. + pub(crate) upsample: Upsample, +} + +/// Per-component geometry for one image at one DCT scale. +#[derive(Clone, Copy, Debug)] +pub(crate) struct ScaledSampling { + components: [ScaledComponent; 4], + /// Sampling factors counted in luma-sized output blocks: each component's + /// factors times its IDCT enlargement. Component planes, MCU strides and + /// upsampling ratios follow these; the entropy decode keeps the coded + /// factors. + pub(crate) effective: SamplingFactors, +} + +impl ScaledSampling { + /// Geometry of an image with `sampling` and full-size `dimensions` + /// decoded at `downscale`. + pub(crate) fn new( + sampling: SamplingFactors, + downscale: DownscaleFactor, + dimensions: (u32, u32), + ) -> Self { + let block_size = downscale.output_block_size(); + let max_h = u32::from(sampling.max_h); + let max_v = u32::from(sampling.max_v); + // jdsample.c: `do_fancy = do_fancy_upsampling && min_DCT_scaled_size > 1`. + let smoothing = block_size > 1; + let mut components = [ScaledComponent { + idct_size: block_size, + h_ratio: 1, + v_ratio: 1, + width: 0, + height: 0, + upsample: Upsample::None, + }; 4]; + let mut effective = [(1u8, 1u8); 4]; + for (index, (h, v)) in sampling.iter().enumerate() { + let (h, v) = (u32::from(h), u32::from(v)); + // jdmaster.c `jpeg_calc_output_dimensions`: grow the IDCT while + // both upsampling ratios stay even. + let mut idct_size = block_size; + while idct_size < 8 + && (max_h * block_size).is_multiple_of(h * idct_size * 2) + && (max_v * block_size).is_multiple_of(v * idct_size * 2) + { + idct_size *= 2; + } + let h_ratio = max_h * block_size / (h * idct_size); + let v_ratio = max_v * block_size / (v * idct_size); + let width = downsampled_extent(dimensions.0, h * idct_size, max_h); + let height = downsampled_extent(dimensions.1, v * idct_size, max_v); + let upsample = match (h_ratio, v_ratio) { + (1, 1) => Upsample::None, + (2, 1) if smoothing && width > 2 => Upsample::FancyH2V1, + (1, 2) if smoothing => Upsample::FancyH1V2, + (2, 2) if smoothing && width > 2 => Upsample::FancyH2V2, + _ => Upsample::Replicate, + }; + components[index] = ScaledComponent { + idct_size, + h_ratio, + v_ratio, + width, + height, + upsample, + }; + effective[index] = ( + effective_factor(h, idct_size, block_size), + effective_factor(v, idct_size, block_size), + ); + } + Self { + components, + effective: SamplingFactors::from_validated_components(&effective[..sampling.len()]), + } + } + + /// Geometry of the component at `index` (declaration and plane order). + pub(crate) fn component(&self, index: usize) -> ScaledComponent { + debug_assert!(index < self.effective.len()); + self.components[index] + } + + /// Whether every component decodes straight to output resolution. + pub(crate) fn is_unsampled(&self) -> bool { + self.components[..self.effective.len()] + .iter() + .all(|component| component.upsample == Upsample::None) + } + + /// Whether this is three-component data whose chroma both use the 4:2:0 + /// smoothing filter against a 2x2 first component, the shape the + /// dedicated 4:2:0 emitters handle. + pub(crate) fn is_fancy_420(&self) -> bool { + self.effective.components() == [(2, 2), (1, 1), (1, 1)] + && self.components[1].upsample == Upsample::FancyH2V2 + && self.components[2].upsample == Upsample::FancyH2V2 + } +} + +/// `jdiv_round_up(image_extent * factor * idct_size, max_factor * DCTSIZE)`. +fn downsampled_extent(image_extent: u32, factor_times_idct: u32, max_factor: u32) -> u32 { + let numerator = u64::from(image_extent) * u64::from(factor_times_idct); + let denominator = u64::from(max_factor) * 8; + u32::try_from(numerator.div_ceil(denominator)).unwrap_or(u32::MAX) +} + +#[expect( + clippy::cast_possible_truncation, + reason = "IDCT enlargement never lifts a factor above the 1..=4 maximum" +)] +fn effective_factor(factor: u32, idct_size: u32, block_size: u32) -> u8 { + (factor * idct_size / block_size) as u8 +} + +#[cfg(test)] +mod tests { + use super::{ScaledSampling, Upsample}; + use crate::info::{DownscaleFactor, SamplingFactors}; + + fn chroma(luma: (u8, u8), downscale: DownscaleFactor, width: u32) -> (u32, u32, u32, Upsample) { + let sampling = SamplingFactors::from_components(&[luma, (1, 1), (1, 1)]).unwrap(); + let chroma = ScaledSampling::new(sampling, downscale, (width, 64)).component(1); + ( + chroma.idct_size, + chroma.h_ratio, + chroma.v_ratio, + chroma.upsample, + ) + } + + #[test] + fn subsampled_chroma_uses_libjpeg_turbo_idct_sizes() { + use DownscaleFactor::{Eighth, Full, Half, Quarter}; + use Upsample::{FancyH1V2, FancyH2V1, FancyH2V2, None, Replicate}; + // 4:2:0 chroma is decoded at twice the luma IDCT size below full. + assert_eq!(chroma((2, 2), Full, 64), (8, 2, 2, FancyH2V2)); + assert_eq!(chroma((2, 2), Half, 64), (8, 1, 1, None)); + assert_eq!(chroma((2, 2), Quarter, 64), (4, 1, 1, None)); + assert_eq!(chroma((2, 2), Eighth, 64), (2, 1, 1, None)); + // 4:2:2 and 4:4:0 keep the luma size; 1/8 replicates. + assert_eq!(chroma((2, 1), Half, 64), (4, 2, 1, FancyH2V1)); + assert_eq!(chroma((2, 1), Eighth, 64), (1, 2, 1, Replicate)); + assert_eq!(chroma((1, 2), Quarter, 64), (2, 1, 2, FancyH1V2)); + assert_eq!(chroma((1, 2), Eighth, 64), (1, 1, 2, Replicate)); + // 4:1:0 chroma grows once, leaving a smoothed 2:1 horizontal step. + assert_eq!(chroma((4, 2), Half, 64), (8, 2, 1, FancyH2V1)); + assert_eq!(chroma((4, 2), Eighth, 64), (2, 2, 1, Replicate)); + // 4:1:1 and 1x4 always replicate. + assert_eq!(chroma((4, 1), Half, 64), (4, 4, 1, Replicate)); + assert_eq!(chroma((1, 4), Full, 64), (8, 1, 4, Replicate)); + } + + #[test] + fn narrow_chroma_replicates_instead_of_smoothing() { + use DownscaleFactor::{Full, Quarter}; + // Four pixels of 4:2:0 or 4:2:2 leave two chroma samples. + assert_eq!(chroma((2, 2), Full, 4).3, Upsample::Replicate); + assert_eq!(chroma((2, 1), Full, 4).3, Upsample::Replicate); + assert_eq!(chroma((2, 2), Full, 5).3, Upsample::FancyH2V2); + // 9 pixels at 1/4 leave ceil(9 * 2 / 16) = 2 chroma samples. + assert_eq!(chroma((2, 1), Quarter, 9).3, Upsample::Replicate); + assert_eq!(chroma((2, 1), Quarter, 17).3, Upsample::FancyH2V1); + // The vertical filter has no width limit. + assert_eq!(chroma((1, 2), Full, 2).3, Upsample::FancyH1V2); + } + + #[test] + fn effective_sampling_counts_luma_sized_blocks() { + let sampling = SamplingFactors::from_components(&[(2, 2), (1, 1), (1, 1)]).unwrap(); + let half = ScaledSampling::new(sampling, DownscaleFactor::Half, (64, 64)); + assert_eq!(half.effective.components(), [(2, 2), (2, 2), (2, 2)]); + assert!(half.is_unsampled()); + assert!(!half.is_fancy_420()); + let full = ScaledSampling::new(sampling, DownscaleFactor::Full, (64, 64)); + assert_eq!(full.effective.components(), [(2, 2), (1, 1), (1, 1)]); + assert!(full.is_fancy_420()); + } +} diff --git a/crates/j2k-jpeg/src/decoder.rs b/crates/j2k-jpeg/src/decoder.rs index 70a4b450c..13217fa45 100644 --- a/crates/j2k-jpeg/src/decoder.rs +++ b/crates/j2k-jpeg/src/decoder.rs @@ -13,9 +13,8 @@ use crate::entropy::progressive::{ }; use crate::entropy::sequential::{ decode_scan_baseline, decode_scan_baseline_rgb, decode_scan_fast_rgb_444, - decode_scan_fast_tile_rgb, decode_scan_fast_tile_rgb_region, - decode_scan_fast_tile_rgb_region_scaled, fast_tile_region_first_decode_mcu, finish_scan, - stripe_region_layout, FastTileRegionScaledRequest, PreparedComponentPlan, PreparedDecodePlan, + decode_scan_fast_tile_rgb, decode_scan_fast_tile_rgb_region, fast_tile_region_first_decode_mcu, + finish_scan, stripe_region_layout, PreparedComponentPlan, PreparedDecodePlan, ResolvedPreparedComponentPlan, }; use crate::entropy::ZIGZAG; @@ -82,7 +81,7 @@ use self::color_convert::{ mod warning_ownership; use self::warning_ownership::{merged_warnings, try_clone_warnings}; mod core_traits; -use self::core_traits::{CroppedWriter, ProgressiveDownscaleWriter}; +use self::core_traits::CroppedWriter; mod lossless_region; use self::lossless_region::{LosslessRegionRequest, LosslessRgbRegionFallback, LosslessRgbaAlpha}; mod scratch; diff --git a/crates/j2k-jpeg/src/decoder/core_traits.rs b/crates/j2k-jpeg/src/decoder/core_traits.rs index 00a487201..2c61f420d 100644 --- a/crates/j2k-jpeg/src/decoder/core_traits.rs +++ b/crates/j2k-jpeg/src/decoder/core_traits.rs @@ -2,14 +2,11 @@ use super::{ ComponentRowWriter, CompressedTransferSyntax, CoreDecodeOutcome, DecodeOutcome, - DecodeRowsError, Decoder, DecoderContext, Downscale, DownscaleFactor, ImageCodec, ImageDecode, - ImageDecodeRows, InterleavedRgbWriter, JpegCodec, JpegError, JpegView, OutputWriter, - PixelFormat, Rect, RowSink, ScratchPool, SofKind, TileBatchDecode, Vec, Warning, -}; -use crate::allocation::{ - checked_add_allocation_bytes, checked_allocation_bytes, checked_allocation_len, - try_reserve_for_len_with_live_budget, try_resize_filled, + DecodeRowsError, Decoder, DecoderContext, Downscale, ImageCodec, ImageDecode, ImageDecodeRows, + InterleavedRgbWriter, JpegCodec, JpegError, JpegView, OutputWriter, PixelFormat, Rect, RowSink, + ScratchPool, SofKind, TileBatchDecode, Vec, Warning, }; +use crate::allocation::{checked_allocation_len, try_reserve_for_len_with_live_budget}; use j2k_core::TileRegionScaledDecodeJob; #[cfg(test)] @@ -260,121 +257,6 @@ pub(super) struct CroppedWriter { pub(super) bottom_row: Vec, } -pub(super) struct ProgressiveDownscaleWriter<'a, W> { - pub(super) inner: &'a mut W, - pub(super) denom: u32, - pub(super) scaled_width: usize, - pub(super) r: Vec, - pub(super) g: Vec, - pub(super) b: Vec, -} - -impl<'a, W> ProgressiveDownscaleWriter<'a, W> { - pub(super) fn new( - inner: &'a mut W, - downscale: DownscaleFactor, - dimensions: (u32, u32), - ) -> Result { - let denom = downscale.denominator(); - let scaled_width = dimensions.0.div_ceil(denom) as usize; - let row_bytes = checked_allocation_len::(scaled_width, 3)?; - let mut live_bytes = 0; - let mut r = Vec::new(); - try_reserve_for_len_with_live_budget(&mut r, scaled_width, &mut live_bytes, row_bytes)?; - r.resize(scaled_width, 0); - let mut g = Vec::new(); - try_reserve_for_len_with_live_budget(&mut g, scaled_width, &mut live_bytes, row_bytes)?; - g.resize(scaled_width, 0); - let mut b = Vec::new(); - try_reserve_for_len_with_live_budget(&mut b, scaled_width, &mut live_bytes, row_bytes)?; - b.resize(scaled_width, 0); - Ok(Self { - inner, - denom, - scaled_width, - r, - g, - b, - }) - } - - pub(super) fn capacity_bytes(&self) -> Result { - let rg = checked_add_allocation_bytes( - checked_allocation_bytes::(self.r.capacity())?, - checked_allocation_bytes::(self.g.capacity())?, - )?; - checked_add_allocation_bytes(rg, checked_allocation_bytes::(self.b.capacity())?) - } - - fn should_emit(&self, y: u32) -> bool { - y.is_multiple_of(self.denom) - } - - #[expect( - clippy::cast_possible_truncation, - reason = "validated JPEG output widths originate as u32 dimensions before slice indexing" - )] - fn sample_row( - src: &[u8], - denom: u32, - width: usize, - dst: &mut Vec, - ) -> Result<(), JpegError> { - try_resize_filled(dst, width, 0)?; - for (x, out) in dst.iter_mut().enumerate() { - let src_x = (x as u32) - .saturating_mul(denom) - .min(src.len().saturating_sub(1) as u32); - *out = src[src_x as usize]; - } - Ok(()) - } -} - -impl OutputWriter for ProgressiveDownscaleWriter<'_, W> { - fn write_rgb_row( - &mut self, - y: u32, - r_row: &[u8], - g_row: &[u8], - b_row: &[u8], - ) -> Result<(), JpegError> { - if !self.should_emit(y) { - return Ok(()); - } - Self::sample_row(r_row, self.denom, self.scaled_width, &mut self.r)?; - Self::sample_row(g_row, self.denom, self.scaled_width, &mut self.g)?; - Self::sample_row(b_row, self.denom, self.scaled_width, &mut self.b)?; - self.inner - .write_rgb_row(y / self.denom, &self.r, &self.g, &self.b) - } - - fn write_ycbcr_row( - &mut self, - y: u32, - y_row: &[u8], - cb_row: &[u8], - cr_row: &[u8], - ) -> Result<(), JpegError> { - if !self.should_emit(y) { - return Ok(()); - } - Self::sample_row(y_row, self.denom, self.scaled_width, &mut self.r)?; - Self::sample_row(cb_row, self.denom, self.scaled_width, &mut self.g)?; - Self::sample_row(cr_row, self.denom, self.scaled_width, &mut self.b)?; - self.inner - .write_ycbcr_row(y / self.denom, &self.r, &self.g, &self.b) - } - - fn write_gray_row(&mut self, y: u32, gray_row: &[u8]) -> Result<(), JpegError> { - if !self.should_emit(y) { - return Ok(()); - } - Self::sample_row(gray_row, self.denom, self.scaled_width, &mut self.r)?; - self.inner.write_gray_row(y / self.denom, &self.r) - } -} - impl OutputWriter for &mut W { fn write_rgb_row( &mut self, diff --git a/crates/j2k-jpeg/src/decoder/core_traits/tests.rs b/crates/j2k-jpeg/src/decoder/core_traits/tests.rs index f453b2785..c5f8363f0 100644 --- a/crates/j2k-jpeg/src/decoder/core_traits/tests.rs +++ b/crates/j2k-jpeg/src/decoder/core_traits/tests.rs @@ -4,9 +4,9 @@ use core::fmt; use j2k_core::{DecodeRowsError, ImageDecode, ImageDecodeRows, TileBatchDecode}; use super::{ - core_outcome, CroppedWriter, Decoder, DecoderContext, Downscale, DownscaleFactor, - InterleavedRgbWriter, JpegCodec, JpegError, OutputWriter, PixelFormat, - ProgressiveDownscaleWriter, Rect, RowSink, ScratchPool, TileRegionScaledDecodeJob, Warning, + core_outcome, CroppedWriter, Decoder, DecoderContext, Downscale, InterleavedRgbWriter, + JpegCodec, JpegError, OutputWriter, PixelFormat, Rect, RowSink, ScratchPool, + TileRegionScaledDecodeJob, Warning, }; const JPEG: &[u8] = j2k_test_support::JPEG_BASELINE_420_16X16; @@ -291,47 +291,6 @@ fn core_row_adapter_preserves_the_original_sink_error_type() { assert!(matches!(error, DecodeRowsError::Sink(SinkStopped))); } -#[test] -fn progressive_downscale_writer_samples_each_output_mode_and_skips_intermediate_rows() { - let mut rows = RecordedRows::default(); - { - let mut writer = ProgressiveDownscaleWriter::new(&mut rows, DownscaleFactor::Half, (5, 4)) - .expect("bounded progressive row scratch"); - assert!(writer.capacity_bytes().expect("row capacity") >= 9); - - writer - .write_rgb_row(1, &[1; 5], &[2; 5], &[3; 5]) - .expect("skipped RGB row"); - writer - .write_rgb_row( - 2, - &[1, 2, 3, 4, 5], - &[6, 7, 8, 9, 10], - &[11, 12, 13, 14, 15], - ) - .expect("sampled RGB row"); - writer - .write_ycbcr_row(0, &[21, 22, 23, 24, 25], &[31; 5], &[41; 5]) - .expect("sampled YCbCr row"); - writer - .write_gray_row(1, &[51; 5]) - .expect("skipped grayscale row"); - writer - .write_gray_row(2, &[51, 52, 53, 54, 55]) - .expect("sampled grayscale row"); - } - - assert_eq!( - rows.rgb, - vec![(1, vec![1, 3, 5], vec![6, 8, 10], vec![11, 13, 15])] - ); - assert_eq!( - rows.ycbcr, - vec![(0, vec![21, 23, 25], vec![31; 3], vec![41; 3])] - ); - assert_eq!(rows.gray, vec![(1, vec![51, 53, 55])]); -} - #[test] fn cropped_interleaved_writer_emits_only_rows_inside_the_source_window() { let inner = RecordedRows { diff --git a/crates/j2k-jpeg/src/decoder/extended12.rs b/crates/j2k-jpeg/src/decoder/extended12.rs index 34e068520..86f15ec0b 100644 --- a/crates/j2k-jpeg/src/decoder/extended12.rs +++ b/crates/j2k-jpeg/src/decoder/extended12.rs @@ -9,6 +9,7 @@ mod planes; mod progressive; mod rgba; mod sampling; +mod scaled; mod sequential; mod state; mod upsample; diff --git a/crates/j2k-jpeg/src/decoder/extended12/progressive.rs b/crates/j2k-jpeg/src/decoder/extended12/progressive.rs index 60a1a9dd0..b429e4530 100644 --- a/crates/j2k-jpeg/src/decoder/extended12/progressive.rs +++ b/crates/j2k-jpeg/src/decoder/extended12/progressive.rs @@ -10,9 +10,11 @@ use super::planes::{dequantize_progressive12_block, ensure_progressive12_coeffic use super::sampling::{ progressive_color_sampling, progressive_four_component_sampling, Extended12ColorSampling, }; +use super::scaled::{needs_scaled_extended12_route, Extended12Color}; use super::writers::{ write_extended12_block_region, Extended12Output, Extended12RgbProjection, Extended12WriteRegion, }; +use crate::color::scaled_sampling::ScaledSampling; mod color444; mod four_component; @@ -82,10 +84,20 @@ impl Decoder<'_> { height: self.info.dimensions.1, }); } + let scaled_route = needs_scaled_extended12_route( + &ScaledSampling::new(plan.sampling, downscale, self.info.dimensions), + downscale, + ); if matches!(output, Extended12Output::Rgb16) { match self.info.color_space { ColorSpace::Rgb => { let sampling = progressive_color_sampling(plan, self.info.sof_kind)?; + if scaled_route { + let color = Extended12Color::Rgb(Extended12RgbProjection::Identity); + return self.decode_extended12_scaled_region_into( + out, stride, roi, downscale, color, + ); + } return match sampling { Extended12ColorSampling::S444 => self .decode_progressive12_color444_region_into( @@ -108,6 +120,12 @@ impl Decoder<'_> { } ColorSpace::YCbCr => { let sampling = progressive_color_sampling(plan, self.info.sof_kind)?; + if scaled_route { + let color = Extended12Color::Rgb(Extended12RgbProjection::YCbCr); + return self.decode_extended12_scaled_region_into( + out, stride, roi, downscale, color, + ); + } return match sampling { Extended12ColorSampling::S444 => self .decode_progressive12_color444_region_into( @@ -130,6 +148,12 @@ impl Decoder<'_> { } ColorSpace::Cmyk | ColorSpace::Ycck => { let sampling = progressive_four_component_sampling(plan, self.info.sof_kind)?; + if scaled_route { + let color = Extended12Color::FourComponent(self.info.color_space); + return self.decode_extended12_scaled_region_into( + out, stride, roi, downscale, color, + ); + } return self.decode_progressive12_four_component_region_into( out, stride, roi, downscale, sampling, ); @@ -142,6 +166,15 @@ impl Decoder<'_> { sof: self.info.sof_kind, }); } + if scaled_route { + return self.decode_extended12_scaled_region_into( + out, + stride, + roi, + downscale, + Extended12Color::Gray(output), + ); + } let output_rect = scaled_rect_covering(roi, downscale)?; let dct_blocks = decode_progressive_dct_blocks(plan, self.bytes, 0)?; diff --git a/crates/j2k-jpeg/src/decoder/extended12/scaled.rs b/crates/j2k-jpeg/src/decoder/extended12/scaled.rs new file mode 100644 index 000000000..4968630b7 --- /dev/null +++ b/crates/j2k-jpeg/src/decoder/extended12/scaled.rs @@ -0,0 +1,470 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 + +//! libjpeg-turbo DCT scaling for 12-bit sequential and progressive images. +//! +//! Each component is inverse-transformed with the reduced IDCT libjpeg-turbo +//! picks for it ([`ScaledSampling`]) into a plane at its scaled resolution; +//! output rows then upsample every plane with libjpeg-turbo's upsampler for +//! that component and convert the color. Full-size decodes take this route +//! only when libjpeg-turbo replicates a narrow 2:1 component instead of +//! smoothing it; every other full-size decode keeps the direct writers. + +use alloc::vec::Vec; + +use super::super::{ + checked_scratch_len, decode_block_with_activity, decode_progressive_dct_blocks, finish_scan, + merged_warnings, scaled_rect_covering, try_clone_warnings, Backend, BitReader, BlockActivity, + CoefficientBlock, ColorSpace, DecodeOutcome, Decoder, DownscaleFactor, JpegError, Rect, + SofKind, +}; +use super::planes::{dequantize_progressive12_block, ensure_progressive12_coefficient_capacities}; +use super::state::Extended12RestartTracker; +use super::writers::{Extended12Output, Extended12RgbProjection}; +use crate::allocation::try_reserve_for_len_with_live_budget; +use crate::color::scaled_sampling::{ScaledComponent, ScaledSampling, Upsample}; +use crate::idct::downscale::{idct_islow_12bit_1x1, idct_islow_12bit_2x2, idct_islow_12bit_4x4}; + +/// How 12-bit component samples become output pixels. +#[derive(Clone, Copy, Debug)] +pub(super) enum Extended12Color { + /// One component written as `Gray16`, or as `Rgb16` with equal channels. + Gray(Extended12Output), + /// Three components written as `Rgb16`. + Rgb(Extended12RgbProjection), + /// Inverted CMYK or YCCK written as `Rgb16`. + FourComponent(ColorSpace), +} + +impl Extended12Color { + fn components(self) -> usize { + match self { + Self::Gray(_) => 1, + Self::Rgb(_) => 3, + Self::FourComponent(_) => 4, + } + } +} + +/// Whether a 12-bit decode needs libjpeg-turbo's scaled geometry: any reduced +/// scale, or a full-size 2:1 component that libjpeg-turbo replicates because +/// it is at most two samples wide. +pub(super) fn needs_scaled_extended12_route( + scaled: &ScaledSampling, + downscale: DownscaleFactor, +) -> bool { + downscale != DownscaleFactor::Full + || (0..scaled.effective.len()).any(|index| { + let component = scaled.component(index); + component.upsample == Upsample::Replicate && component.h_ratio == 2 + }) +} + +/// A component plane at its scaled resolution, padded to whole blocks. +struct ScaledPlane { + samples: Vec, + stride: usize, + component: ScaledComponent, +} + +impl ScaledPlane { + fn row(&self, y: usize) -> &[u16] { + let y = y.min(self.component.height.saturating_sub(1) as usize); + let start = y * self.stride; + &self.samples[start..start + self.component.width as usize] + } + + fn deposit(&mut self, x: usize, y: usize, size: usize, block: &[u16]) { + for row in 0..size { + let dst = (y + row) * self.stride + x; + self.samples[dst..dst + size].copy_from_slice(&block[row * size..(row + 1) * size]); + } + } +} + +/// At most four component planes, held inline so only sample storage is +/// charged to the decode budget. +struct ScaledPlanes { + planes: [ScaledPlane; 4], + count: usize, +} + +impl ScaledPlanes { + fn as_slice(&self) -> &[ScaledPlane] { + &self.planes[..self.count] + } + + fn get_mut(&mut self, index: usize) -> Option<&mut ScaledPlane> { + self.planes[..self.count].get_mut(index) + } + + fn sample_bytes(&self) -> Result { + self.as_slice().iter().try_fold(0usize, |total, plane| { + let bytes = checked_scratch_len(&[plane.samples.capacity(), 2])?; + checked_scratch_len(&[1, total]).map(|total| total.saturating_add(bytes)) + }) + } +} + +/// Allocates one plane per component, `blocks[i]` (columns, rows) blocks of +/// the component's IDCT size, charging `cap`. +fn allocate_scaled_planes( + scaled: &ScaledSampling, + blocks: &[(usize, usize)], + initial_live_bytes: usize, + cap: usize, +) -> Result { + let mut live_bytes = initial_live_bytes; + let mut planes: [ScaledPlane; 4] = core::array::from_fn(|index| ScaledPlane { + samples: Vec::new(), + stride: 0, + component: scaled.component(index.min(blocks.len().saturating_sub(1))), + }); + for (index, &(cols, rows)) in blocks.iter().enumerate() { + let component = scaled.component(index); + let size = component.idct_size as usize; + let stride = checked_scratch_len(&[cols, size])?; + let len = checked_scratch_len(&[stride, rows, size])?; + let mut samples = Vec::new(); + try_reserve_for_len_with_live_budget(&mut samples, len, &mut live_bytes, cap)?; + samples.resize(len, 0); + planes[index] = ScaledPlane { + samples, + stride, + component, + }; + } + Ok(ScaledPlanes { + planes, + count: blocks.len(), + }) +} + +/// Reduced IDCT of one dequantized block into `pixels` (`size * size` +/// samples); `dc_only` lets the full-size transform take its DC shortcut. +fn idct_12bit_block( + backend: Backend, + coefficients: &[i16; 64], + dc_only: bool, + size: usize, + pixels: &mut [u16; 64], +) { + match size { + 8 if dc_only => pixels.fill(crate::idct::idct_islow_12bit_dc_only_sample( + coefficients[0], + )), + 8 => backend.idct_12bit(coefficients, pixels), + 4 => { + let mut block = [0u16; 16]; + idct_islow_12bit_4x4(coefficients, &mut block); + pixels[..16].copy_from_slice(&block); + } + 2 => { + let mut block = [0u16; 4]; + idct_islow_12bit_2x2(coefficients, &mut block); + pixels[..4].copy_from_slice(&block); + } + _ => pixels[0] = idct_islow_12bit_1x1(coefficients), + } +} + +impl Decoder<'_> { + /// Decodes `roi` of a 12-bit image at `downscale` with libjpeg-turbo's + /// per-component reduced IDCTs and upsamplers. + pub(super) fn decode_extended12_scaled_region_into( + &self, + out: &mut [u8], + stride: usize, + roi: Rect, + downscale: DownscaleFactor, + color: Extended12Color, + ) -> Result { + let output_rect = scaled_rect_covering(roi, downscale)?; + let (planes, warnings, cap) = if self.info.sof_kind == SofKind::Progressive12 { + let planes = self.render_progressive12_scaled_planes(downscale, color)?; + let cap = self + .progressive_plan + .as_ref() + .map_or(0, |plan| plan.scratch_bytes); + (planes, try_clone_warnings(&self.warnings)?, cap) + } else { + let (planes, scan_warnings) = self.decode_extended12_scaled_planes(downscale, color)?; + let warnings = merged_warnings(&self.warnings, scan_warnings)?; + (planes, warnings, self.plan.scratch_bytes) + }; + let live_bytes = planes.sample_bytes()?; + let write = ScaledWrite { + output_rect, + output_width: self.info.dimensions.0.div_ceil(downscale.denominator()) as usize, + live_bytes, + cap, + }; + write_scaled_planes_region(out, stride, write, planes.as_slice(), color)?; + Ok(DecodeOutcome { + decoded: roi, + warnings, + }) + } + + fn decode_extended12_scaled_planes( + &self, + downscale: DownscaleFactor, + color: Extended12Color, + ) -> Result<(ScaledPlanes, Vec), JpegError> { + let plan = &self.plan; + let sof = self.info.sof_kind; + if plan.components.len() != color.components() { + return Err(JpegError::NotImplemented { sof }); + } + let scaled = ScaledSampling::new(plan.sampling, downscale, plan.dimensions); + let max_h = u32::from(plan.sampling.max_h); + let max_v = u32::from(plan.sampling.max_v); + let mcu_cols = plan.dimensions.0.div_ceil(max_h * 8); + let mcu_rows = plan.dimensions.1.div_ceil(max_v * 8); + let blocks: Vec<(usize, usize)> = plan + .sampling + .iter() + .map(|(h, v)| { + ( + mcu_cols as usize * usize::from(h), + mcu_rows as usize * usize::from(v), + ) + }) + .collect(); + let mut planes = allocate_scaled_planes(&scaled, &blocks, 0, plan.scratch_bytes)?; + + let mut br = BitReader::new(&self.bytes[plan.scan_offset..]); + let mut prev_dc = [0i32; 4]; + let mut coeff = CoefficientBlock::default(); + let mut pixels = [0u16; 64]; + let mut restart_tracker = + Extended12RestartTracker::new(plan.restart_interval, mcu_cols * mcu_rows); + for mcu_y in 0..mcu_rows { + for mcu_x in 0..mcu_cols { + if restart_tracker.begin_mcu(&mut br, mcu_y * mcu_cols + mcu_x)? { + prev_dc.fill(0); + } + for component in &plan.components { + let index = component.output_index; + let resolved = plan.resolve_component(component)?; + let plane = planes + .get_mut(index) + .ok_or(JpegError::NotImplemented { sof })?; + let size = plane.component.idct_size as usize; + let (h, v) = (u32::from(component.h), u32::from(component.v)); + for by in 0..v { + for bx in 0..h { + let activity = decode_block_with_activity( + &mut br, + resolved.dc_table, + resolved.ac_table, + &mut prev_dc[index], + resolved.quant, + &mut coeff, + )?; + idct_12bit_block( + self.backend, + coeff.coefficients(), + activity == BlockActivity::DcOnly, + size, + &mut pixels, + ); + plane.deposit( + (mcu_x * h + bx) as usize * size, + (mcu_y * v + by) as usize * size, + size, + &pixels, + ); + } + } + } + restart_tracker.finish_mcu(); + } + } + let scan_warnings = finish_scan(&mut br, true)?; + Ok((planes, scan_warnings)) + } + + fn render_progressive12_scaled_planes( + &self, + downscale: DownscaleFactor, + color: Extended12Color, + ) -> Result { + let sof = self.info.sof_kind; + let plan = self + .progressive_plan + .as_ref() + .ok_or(JpegError::NotImplemented { sof })?; + if plan.components.len() != color.components() { + return Err(JpegError::NotImplemented { sof }); + } + let scaled = ScaledSampling::new(plan.sampling, downscale, plan.dimensions); + let dct_blocks = decode_progressive_dct_blocks(plan, self.bytes, 0)?; + ensure_progressive12_coefficient_capacities(&dct_blocks, plan.scratch_bytes)?; + let mut blocks = [(0usize, 0usize); 4]; + for component in &plan.components { + let slot = blocks + .get_mut(component.output_index) + .ok_or(JpegError::NotImplemented { sof })?; + *slot = (component.block_cols as usize, component.block_rows as usize); + } + let mut planes = allocate_scaled_planes( + &scaled, + &blocks[..plan.components.len()], + dct_blocks.capacity_bytes()?, + plan.scratch_bytes, + )?; + let mut dequant = [0i16; 64]; + let mut pixels = [0u16; 64]; + for (component, coeffs) in plan.components.iter().zip(&dct_blocks.quantized) { + let plane = planes + .get_mut(component.output_index) + .ok_or(JpegError::NotImplemented { sof })?; + let size = plane.component.idct_size as usize; + let cols = component.block_cols as usize; + for (block_index, block) in coeffs.iter().enumerate() { + dequantize_progressive12_block(block, &component.quant, &mut dequant); + let dc_only = dequant[1..].iter().all(|&coefficient| coefficient == 0); + idct_12bit_block(self.backend, &dequant, dc_only, size, &mut pixels); + plane.deposit( + (block_index % cols) * size, + (block_index / cols) * size, + size, + &pixels, + ); + } + } + Ok(planes) + } +} + +/// One component's output row `y`, upsampled across the full scaled width. +fn upsample_row(plane: &ScaledPlane, y: usize, out: &mut [u16]) { + let ScaledComponent { + h_ratio, + v_ratio, + upsample, + .. + } = plane.component; + match upsample { + Upsample::None => out.copy_from_slice(&plane.row(y)[..out.len()]), + Upsample::FancyH2V1 => { + let row = plane.row(y); + for (x, dst) in out.iter_mut().enumerate() { + *dst = super::upsample::upsample_extended12_h2v1_at(row, x); + } + } + Upsample::FancyH1V2 => { + // h1v2_fancy_upsample: 3/4 nearer row + 1/4 further row, biased + // +1 for the upper output row and +2 for the lower. + let sample_y = y / 2; + let curr = plane.row(sample_y); + let (near, bias) = if y.is_multiple_of(2) { + (plane.row(sample_y.saturating_sub(1)), 1) + } else { + (plane.row(sample_y + 1), 2) + }; + for (x, dst) in out.iter_mut().enumerate() { + let sum = 3 * u32::from(curr[x]) + u32::from(near[x]) + bias; + *dst = u16::try_from(sum >> 2).expect("weighted mean of 12-bit samples fits u16"); + } + } + Upsample::FancyH2V2 => { + let sample_y = y / 2; + let prev = plane.row(sample_y.saturating_sub(1)); + let curr = plane.row(sample_y); + let next = plane.row(sample_y + 1); + for (x, dst) in out.iter_mut().enumerate() { + *dst = super::upsample::upsample_h2v2_u16_rows_at(prev, curr, next, x, y % 2 == 1); + } + } + Upsample::Replicate => { + let row = plane.row(y / v_ratio as usize); + let h_ratio = h_ratio as usize; + for (x, dst) in out.iter_mut().enumerate() { + *dst = row[(x / h_ratio).min(row.len() - 1)]; + } + } + } +} + +/// Output geometry and memory budget for writing scaled planes. +#[derive(Clone, Copy)] +struct ScaledWrite { + output_rect: Rect, + /// Width of the scaled image; every upsampled row spans it. + output_width: usize, + /// Bytes already live (the planes) and the cap they share with the rows. + live_bytes: usize, + cap: usize, +} + +fn write_scaled_planes_region( + out: &mut [u8], + stride: usize, + write: ScaledWrite, + planes: &[ScaledPlane], + color: Extended12Color, +) -> Result<(), JpegError> { + let ScaledWrite { + output_rect, + output_width, + mut live_bytes, + cap, + } = write; + let x0 = output_rect.x as usize; + let x1 = x0 + output_rect.w as usize; + let mut rows: [Vec; 4] = Default::default(); + for row in &mut rows[..planes.len()] { + try_reserve_for_len_with_live_budget(row, output_width, &mut live_bytes, cap)?; + row.resize(output_width, 0u16); + } + for y in output_rect.y..output_rect.y + output_rect.h { + for (plane, row) in planes.iter().zip(rows.iter_mut()) { + upsample_row(plane, y as usize, row); + } + let dst_row = &mut out[(y - output_rect.y) as usize * stride..]; + for (dst_x, x) in (x0..x1).enumerate() { + match color { + Extended12Color::Gray(Extended12Output::Gray16) => { + dst_row[dst_x * 2..dst_x * 2 + 2].copy_from_slice(&rows[0][x].to_le_bytes()); + } + Extended12Color::Gray(Extended12Output::Rgb16) => { + let sample = rows[0][x]; + write_rgb16_le( + &mut dst_row[dst_x * 6..dst_x * 6 + 6], + (sample, sample, sample), + ); + } + Extended12Color::Rgb(projection) => { + let (r, g, b) = match projection { + Extended12RgbProjection::Identity => (rows[0][x], rows[1][x], rows[2][x]), + Extended12RgbProjection::YCbCr => crate::color::ycbcr::ycbcr12_to_rgb16( + rows[0][x], rows[1][x], rows[2][x], + ), + }; + write_rgb16_le(&mut dst_row[dst_x * 6..dst_x * 6 + 6], (r, g, b)); + } + Extended12Color::FourComponent(color_space) => { + let samples = (rows[0][x], rows[1][x], rows[2][x], rows[3][x]); + let rgb = if color_space == ColorSpace::Cmyk { + crate::color::cmyk::inverted_cmyk12_to_rgb16( + samples.0, samples.1, samples.2, samples.3, + ) + } else { + crate::color::cmyk::ycck12_to_rgb16( + samples.0, samples.1, samples.2, samples.3, + ) + }; + write_rgb16_le(&mut dst_row[dst_x * 6..dst_x * 6 + 6], rgb); + } + } + } + } + Ok(()) +} + +fn write_rgb16_le(dst: &mut [u8], (r, g, b): (u16, u16, u16)) { + dst[0..2].copy_from_slice(&r.to_le_bytes()); + dst[2..4].copy_from_slice(&g.to_le_bytes()); + dst[4..6].copy_from_slice(&b.to_le_bytes()); +} diff --git a/crates/j2k-jpeg/src/decoder/extended12/sequential.rs b/crates/j2k-jpeg/src/decoder/extended12/sequential.rs index ddc2e886a..ce64bdea1 100644 --- a/crates/j2k-jpeg/src/decoder/extended12/sequential.rs +++ b/crates/j2k-jpeg/src/decoder/extended12/sequential.rs @@ -9,10 +9,12 @@ use super::super::{ use super::sampling::{ extended12_color_sampling, extended12_four_component_sampling, Extended12ColorSampling, }; +use super::scaled::{needs_scaled_extended12_route, Extended12Color}; use super::state::{decode_extended12_block_pixels, Extended12RestartTracker}; use super::writers::{ write_extended12_block_region, Extended12Output, Extended12RgbProjection, Extended12WriteRegion, }; +use crate::color::scaled_sampling::ScaledSampling; mod color444; mod four_component; @@ -121,10 +123,20 @@ impl Decoder<'_> { height: self.info.dimensions.1, }); } + let scaled_route = needs_scaled_extended12_route( + &ScaledSampling::new(self.plan.sampling, downscale, self.info.dimensions), + downscale, + ); if matches!(output, Extended12Output::Rgb16) { match self.info.color_space { ColorSpace::Rgb => { let sampling = extended12_color_sampling(&self.plan, self.info.sof_kind)?; + if scaled_route { + let color = Extended12Color::Rgb(Extended12RgbProjection::Identity); + return self.decode_extended12_scaled_region_into( + out, stride, roi, downscale, color, + ); + } return match sampling { Extended12ColorSampling::S444 => self .decode_extended12_color444_region_into( @@ -147,6 +159,12 @@ impl Decoder<'_> { } ColorSpace::YCbCr => { let sampling = extended12_color_sampling(&self.plan, self.info.sof_kind)?; + if scaled_route { + let color = Extended12Color::Rgb(Extended12RgbProjection::YCbCr); + return self.decode_extended12_scaled_region_into( + out, stride, roi, downscale, color, + ); + } return match sampling { Extended12ColorSampling::S444 => self .decode_extended12_color444_region_into( @@ -170,6 +188,12 @@ impl Decoder<'_> { ColorSpace::Cmyk | ColorSpace::Ycck => { let sampling = extended12_four_component_sampling(&self.plan, self.info.sof_kind)?; + if scaled_route { + let color = Extended12Color::FourComponent(self.info.color_space); + return self.decode_extended12_scaled_region_into( + out, stride, roi, downscale, color, + ); + } return match sampling { Extended12ColorSampling::S444 => self .decode_extended12_four_component444_region_into( @@ -189,6 +213,15 @@ impl Decoder<'_> { sof: self.info.sof_kind, }); } + if scaled_route { + return self.decode_extended12_scaled_region_into( + out, + stride, + roi, + downscale, + Extended12Color::Gray(output), + ); + } let output_rect = scaled_rect_covering(roi, downscale)?; let scan_bytes = &self.bytes[self.plan.scan_offset..]; diff --git a/crates/j2k-jpeg/src/decoder/extended12/writers.rs b/crates/j2k-jpeg/src/decoder/extended12/writers.rs index 0c6e02819..d1a98b7ed 100644 --- a/crates/j2k-jpeg/src/decoder/extended12/writers.rs +++ b/crates/j2k-jpeg/src/decoder/extended12/writers.rs @@ -464,11 +464,23 @@ fn full_block_rows( let rect = region.output_rect; let rows = block_output_span(rect.y, rect.h, 1, height, y0, (y0 + 8).min(height)); let cols = block_output_span(rect.x, rect.w, 1, width, x0, (x0 + 8).min(width)); - let (src_col, dst_col, len) = ( - (cols.start - x0) as usize, - (cols.start - rect.x) as usize, - cols.len(), - ); + // A block outside the output rectangle has an empty span clamped to the + // rectangle's edge, which may lie before the block: emit nothing rather + // than derive offsets from it. + let rows = if cols.is_empty() { + rows.start..rows.start + } else { + rows + }; + let (src_col, dst_col, len) = if cols.is_empty() { + (0, 0, 0) + } else { + ( + (cols.start - x0) as usize, + (cols.start - rect.x) as usize, + cols.len(), + ) + }; rows.map(move |output_y| { ( (output_y - y0) as usize, diff --git a/crates/j2k-jpeg/src/decoder/routing/dispatch.rs b/crates/j2k-jpeg/src/decoder/routing/dispatch.rs index 616fd0942..c5a36027f 100644 --- a/crates/j2k-jpeg/src/decoder/routing/dispatch.rs +++ b/crates/j2k-jpeg/src/decoder/routing/dispatch.rs @@ -3,10 +3,9 @@ //! Output-format dispatch after routing has validated geometry and scratch. use super::super::{ - decode_scan_fast_tile_rgb_region, decode_scan_fast_tile_rgb_region_scaled, - fast_tile_region_first_decode_mcu, merged_warnings, CroppedWriter, DecodeOutcome, Decoder, - DownscaleFactor, FastTileRegionScaledRequest, Gray8Writer, JpegError, OutputFormat, Rect, - Rgb8Writer, Rgba8Writer, ScratchPool, SofKind, + decode_scan_fast_tile_rgb_region, fast_tile_region_first_decode_mcu, merged_warnings, + CroppedWriter, DecodeOutcome, Decoder, DownscaleFactor, Gray8Writer, JpegError, OutputFormat, + Rect, Rgb8Writer, Rgba8Writer, ScratchPool, SofKind, }; use crate::allocation::checked_add_allocation_bytes; @@ -256,34 +255,8 @@ impl Decoder<'_> { warnings: merged_warnings(&self.warnings, scan_warnings)?, }); } - if matches!(fmt, OutputFormat::Rgb8Scaled { .. }) - && self.progressive_plan.is_none() - && self.plan.matches_fast_tile_shape() - { - let mut writer = Rgb8Writer::new_with_backend(out, stride, output_rect.w, self.backend); - let scan_bytes = &self.bytes[self.plan.scan_offset..]; - let checkpoint = self.checkpoint_for_mcu( - scan_bytes, - fast_tile_region_first_decode_mcu(&self.plan, output_rect, downscale), - checked_add_allocation_bytes(external_live_bytes, pool.retained_bytes())?, - )?; - let scan_warnings = decode_scan_fast_tile_rgb_region_scaled( - &self.plan, - self.backend, - scan_bytes, - pool, - &mut writer, - FastTileRegionScaledRequest { - roi: output_rect, - downscale, - checkpoint: checkpoint.as_ref(), - }, - )?; - return Ok(DecodeOutcome { - decoded: output_rect, - warnings: merged_warnings(&self.warnings, scan_warnings)?, - }); - } + // Scaled 4:2:0 takes the generic route: libjpeg-turbo decodes its + // chroma with a larger reduced IDCT, so no 4:2:0 upsampling remains. let base = Rgb8Writer::new_with_backend(out, stride, output_rect.w, self.backend); let (source_x0, source_width) = self.source_window_for_output_rect(downscale, output_rect); let mut writer = CroppedWriter::new(base, output_rect, source_x0, source_width)?; diff --git a/crates/j2k-jpeg/src/decoder/scratch.rs b/crates/j2k-jpeg/src/decoder/scratch.rs index 5c0c85734..e4c63cb5f 100644 --- a/crates/j2k-jpeg/src/decoder/scratch.rs +++ b/crates/j2k-jpeg/src/decoder/scratch.rs @@ -239,7 +239,16 @@ pub(super) fn compute_extended12_planes_scratch_bytes( ) -> Result { let mcu_cols = width.div_ceil(u32::from(sampling.max_h) * 8) as usize; let mcu_rows = height.div_ceil(u32::from(sampling.max_v) * 8) as usize; - let mut total = 0usize; + // The scaled writer holds one u16 output row per component while the + // planes remain live. Full-resolution widths bound every decode scale. + let mut total = checked_usize_product( + &[ + width as usize, + components.len(), + core::mem::size_of::(), + ], + cap, + )?; for component in components { let stride = checked_usize_product(&[mcu_cols, usize::from(component.h), 8], cap)?; let rows = checked_usize_product(&[mcu_rows, usize::from(component.v), 8], cap)?; diff --git a/crates/j2k-jpeg/src/decoder/sequential.rs b/crates/j2k-jpeg/src/decoder/sequential.rs index a9ed2f27a..95b7c1f8d 100644 --- a/crates/j2k-jpeg/src/decoder/sequential.rs +++ b/crates/j2k-jpeg/src/decoder/sequential.rs @@ -10,7 +10,7 @@ use super::{ decode_scan_fast_rgb_444, decode_scan_fast_tile_rgb, emit_decode_scan_profile, jpeg_profile_stages_enabled, merged_warnings, scaled_dimensions, scaled_rect_covering, stripe_region_layout, DecodeOutcome, Decoder, DeviceCheckpoint, DownscaleFactor, Instant, - InterleavedRgbWriter, JpegError, OutputWriter, ProgressiveDownscaleWriter, Rect, ScratchPool, + InterleavedRgbWriter, JpegError, OutputWriter, Rect, ScratchPool, CPU_ROI_CHECKPOINT_CADENCE_MCUS, CPU_ROI_CHECKPOINT_MIN_TARGET_MCUS, }; @@ -143,26 +143,16 @@ impl Decoder<'_> { if let Some(plan) = &self.progressive_plan { let scan_start = profile_enabled.then(Instant::now); let detached_sink_bytes = pool.detached_sink_bytes(); - let scan_warnings = if downscale == DownscaleFactor::Full { - let external_live_bytes = - checked_live_workspace_bytes(detached_sink_bytes, 0, plan.scratch_bytes)?; - decode_progressive(plan, self.backend, self.bytes, writer, external_live_bytes)? - } else { - let mut scaled = - ProgressiveDownscaleWriter::new(writer, downscale, self.info.dimensions)?; - let external_live_bytes = checked_live_workspace_bytes( - detached_sink_bytes, - scaled.capacity_bytes()?, - plan.scratch_bytes, - )?; - decode_progressive( - plan, - self.backend, - self.bytes, - &mut scaled, - external_live_bytes, - )? - }; + let external_live_bytes = + checked_live_workspace_bytes(detached_sink_bytes, 0, plan.scratch_bytes)?; + let scan_warnings = decode_progressive( + plan, + self.backend, + self.bytes, + writer, + external_live_bytes, + downscale, + )?; if let Some(start) = scan_start { emit_decode_scan_profile( "progressive", @@ -216,26 +206,16 @@ impl Decoder<'_> { if let Some(plan) = &self.progressive_plan { let scan_start = profile_enabled.then(Instant::now); let detached_sink_bytes = pool.detached_sink_bytes(); - let scan_warnings = if downscale == DownscaleFactor::Full { - let external_live_bytes = - checked_live_workspace_bytes(detached_sink_bytes, 0, plan.scratch_bytes)?; - decode_progressive(plan, self.backend, self.bytes, writer, external_live_bytes)? - } else { - let mut scaled = - ProgressiveDownscaleWriter::new(writer, downscale, self.info.dimensions)?; - let external_live_bytes = checked_live_workspace_bytes( - detached_sink_bytes, - scaled.capacity_bytes()?, - plan.scratch_bytes, - )?; - decode_progressive( - plan, - self.backend, - self.bytes, - &mut scaled, - external_live_bytes, - )? - }; + let external_live_bytes = + checked_live_workspace_bytes(detached_sink_bytes, 0, plan.scratch_bytes)?; + let scan_warnings = decode_progressive( + plan, + self.backend, + self.bytes, + writer, + external_live_bytes, + downscale, + )?; if let Some(start) = scan_start { emit_decode_scan_profile( "progressive_rgb", diff --git a/crates/j2k-jpeg/src/entropy/block.rs b/crates/j2k-jpeg/src/entropy/block.rs index ee9de6cbd..890d51d51 100644 --- a/crates/j2k-jpeg/src/entropy/block.rs +++ b/crates/j2k-jpeg/src/entropy/block.rs @@ -32,28 +32,6 @@ pub(crate) enum BlockActivity { General, } -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(crate) enum ReducedIdctCoefficients { - Half, - Quarter, -} - -impl ReducedIdctCoefficients { - #[expect( - clippy::inline_always, - reason = "measured entropy-block hot path requires cross-helper inlining" - )] - #[inline(always)] - fn keeps(self, natural_idx: usize) -> bool { - let row = natural_idx >> 3; - let col = natural_idx & 7; - match self { - Self::Half => row != 4 && col != 4, - Self::Quarter => !matches!(row, 2 | 4 | 6) && !matches!(col, 2 | 4 | 6), - } - } -} - #[derive(Debug, Clone)] pub(crate) struct CoefficientBlock { coeffs: [i16; 64], @@ -304,65 +282,6 @@ pub(crate) fn decode_block_with_dc_status( Ok(dc_only) } -#[expect( - clippy::inline_always, - reason = "measured entropy-block hot path requires cross-helper inlining" -)] -#[inline(always)] -pub(crate) fn decode_block_for_reduced_idct( - br: &mut BitReader<'_>, - dc_table: DcHuffmanTable<'_>, - ac_table: AcHuffmanTable<'_>, - prev_dc: &mut i32, - quant: &[u16; 64], - block: &mut CoefficientBlock, - keep: ReducedIdctCoefficients, -) -> Result { - block.clear_touched(); - - let diff = dc_table.decode_fast_dc(br)?; - *prev_dc = prev_dc.wrapping_add(diff); - let dc_dequant = (*prev_dc).wrapping_mul(i32::from(quant[0])); - block.store_dc(clamp_i16(dc_dequant)); - - let mut dc_only_for_reduced_idct = true; - drive_ac_fast::(br, ac_table, |k, ac| { - let natural_idx = ZIGZAG[k] as usize; - if keep.keeps(natural_idx) { - let value = ac_decoded_value(ac); - let dequant = value.wrapping_mul(i32::from(quant[k])); - block.store(natural_idx, clamp_i16(dequant)); - dc_only_for_reduced_idct = false; - } - Ok(()) - })?; - Ok(dc_only_for_reduced_idct) -} - -#[expect( - clippy::inline_always, - reason = "measured entropy-block hot path requires cross-helper inlining" -)] -#[inline(always)] -pub(crate) fn decode_block_for_1x1_idct( - br: &mut BitReader<'_>, - dc_table: DcHuffmanTable<'_>, - ac_table: AcHuffmanTable<'_>, - prev_dc: &mut i32, - quant: &[u16; 64], - block: &mut CoefficientBlock, -) -> Result<(), JpegError> { - block.clear_touched(); - - let diff = dc_table.decode_fast_dc(br)?; - *prev_dc = prev_dc.wrapping_add(diff); - let dc_dequant = (*prev_dc).wrapping_mul(i32::from(quant[0])); - block.store_dc(clamp_i16(dc_dequant)); - - drive_ac_fast::(br, ac_table, |_, _| Ok(()))?; - Ok(()) -} - #[expect( clippy::inline_always, reason = "measured entropy-block hot path requires cross-helper inlining" @@ -633,112 +552,6 @@ mod tests { assert_eq!(dc_status_reader.snapshot(), activity_reader.snapshot()); } - #[test] - fn reduced_idct_decoder_keeps_only_coefficients_read_by_scale() { - let dc = trivial_dc_table(); - let raw = RawHuffmanTable { - bits: [0, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - values: HuffmanValues::from_slice(&[0x41, 0x00]), - }; - let ac = HuffmanTable::from_raw(&raw, HuffmanTableRole::Ac).unwrap(); - let bytes = [0b0001_0100u8, 0, 0, 0]; - let quant = [1u16; 64]; - let ignored_by_quarter = ZIGZAG[5] as usize; - assert_eq!(ignored_by_quarter, 2); - - let mut full_reader = BitReader::new(&bytes); - let mut quarter_reader = BitReader::new(&bytes); - let mut half_reader = BitReader::new(&bytes); - let mut full_prev_dc = 0i32; - let mut quarter_prev_dc = 0i32; - let mut half_prev_dc = 0i32; - let mut full_block = CoefficientBlock::default(); - let mut quarter_block = CoefficientBlock::default(); - let mut half_block = CoefficientBlock::default(); - - decode_block_with_activity( - &mut full_reader, - dc.dc().unwrap(), - ac.ac().unwrap(), - &mut full_prev_dc, - &quant, - &mut full_block, - ) - .unwrap(); - let quarter_dc_only = decode_block_for_reduced_idct( - &mut quarter_reader, - dc.dc().unwrap(), - ac.ac().unwrap(), - &mut quarter_prev_dc, - &quant, - &mut quarter_block, - ReducedIdctCoefficients::Quarter, - ) - .unwrap(); - let half_dc_only = decode_block_for_reduced_idct( - &mut half_reader, - dc.dc().unwrap(), - ac.ac().unwrap(), - &mut half_prev_dc, - &quant, - &mut half_block, - ReducedIdctCoefficients::Half, - ) - .unwrap(); - - assert_eq!(quarter_prev_dc, full_prev_dc); - assert_eq!(half_prev_dc, full_prev_dc); - assert_eq!(quarter_reader.snapshot(), full_reader.snapshot()); - assert_eq!(half_reader.snapshot(), full_reader.snapshot()); - assert_eq!(full_block.coefficients()[ignored_by_quarter], 1); - assert_eq!(quarter_block.coefficients()[ignored_by_quarter], 0); - assert_eq!(half_block.coefficients()[ignored_by_quarter], 1); - assert!(quarter_dc_only); - assert!(!half_dc_only); - } - - #[test] - fn one_by_one_idct_decoder_keeps_dc_and_skips_ac_values() { - let dc = trivial_dc_table(); - let raw = RawHuffmanTable { - bits: [0, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0], - values: HuffmanValues::from_slice(&[0x01, 0x00]), - }; - let ac = HuffmanTable::from_raw(&raw, HuffmanTableRole::Ac).unwrap(); - let bytes = [0b0001_0100u8, 0, 0, 0]; - let quant = [1u16; 64]; - let mut full_reader = BitReader::new(&bytes); - let mut one_by_one_reader = BitReader::new(&bytes); - let mut full_prev_dc = 0i32; - let mut one_by_one_prev_dc = 0i32; - let mut full_block = CoefficientBlock::default(); - let mut one_by_one_block = CoefficientBlock::default(); - - decode_block_with_activity( - &mut full_reader, - dc.dc().unwrap(), - ac.ac().unwrap(), - &mut full_prev_dc, - &quant, - &mut full_block, - ) - .unwrap(); - decode_block_for_1x1_idct( - &mut one_by_one_reader, - dc.dc().unwrap(), - ac.ac().unwrap(), - &mut one_by_one_prev_dc, - &quant, - &mut one_by_one_block, - ) - .unwrap(); - - assert_eq!(one_by_one_prev_dc, full_prev_dc); - assert_eq!(one_by_one_reader.snapshot(), full_reader.snapshot()); - assert_eq!(one_by_one_block.dc_coeff(), full_block.dc_coeff()); - assert!(one_by_one_block.coefficients()[1..].iter().all(|&c| c == 0)); - } - #[test] fn skip_block_consumes_stream_and_updates_dc_like_decode() { let dc_raw = RawHuffmanTable { diff --git a/crates/j2k-jpeg/src/entropy/progressive.rs b/crates/j2k-jpeg/src/entropy/progressive.rs index fa5fd8154..1f51fc32d 100644 --- a/crates/j2k-jpeg/src/entropy/progressive.rs +++ b/crates/j2k-jpeg/src/entropy/progressive.rs @@ -6,7 +6,9 @@ use alloc::vec::Vec; use crate::backend::Backend; +use crate::color::scaled_sampling::ScaledSampling; use crate::error::{JpegError, Warning}; +use crate::info::DownscaleFactor; use crate::output::OutputWriter; mod allocation; @@ -25,13 +27,22 @@ pub(crate) use self::scan::decode_progressive_dct_blocks; use self::allocation::{checked_phase_capacity, component_image_capacity_bytes}; use self::render::{emit_component_images, render_component_images}; +/// Decodes a progressive image at `downscale`, applying libjpeg-turbo's +/// reduced IDCTs to the completed coefficients. pub(crate) fn decode_progressive( plan: &PreparedProgressivePlan, backend: Backend, bytes: &[u8], writer: &mut W, external_live_bytes: usize, + downscale: DownscaleFactor, ) -> Result, JpegError> { + let scaled = ScaledSampling::new(plan.sampling, downscale, plan.dimensions); + let denominator = downscale.denominator(); + let output_dimensions = ( + plan.dimensions.0.div_ceil(denominator), + plan.dimensions.1.div_ceil(denominator), + ); let dct_blocks = decode_progressive_dct_blocks(plan, bytes, external_live_bytes)?; let coefficient_live_bytes = checked_phase_capacity( external_live_bytes, @@ -39,14 +50,22 @@ pub(crate) fn decode_progressive( plan.scratch_bytes, )?; let ProgressiveDctBlocks { quantized: coeffs } = dct_blocks; - let images = render_component_images(plan, backend, &coeffs, coefficient_live_bytes)?; + let images = render_component_images(plan, backend, &scaled, &coeffs, coefficient_live_bytes)?; drop(coeffs); let image_live_bytes = checked_phase_capacity( external_live_bytes, component_image_capacity_bytes(images.capacity(), &images)?, plan.scratch_bytes, )?; - emit_component_images(plan, backend, &images, image_live_bytes, writer)?; + emit_component_images( + plan, + backend, + &scaled, + output_dimensions, + &images, + image_live_bytes, + writer, + )?; Ok(Vec::new()) } diff --git a/crates/j2k-jpeg/src/entropy/progressive/allocation.rs b/crates/j2k-jpeg/src/entropy/progressive/allocation.rs index 234abc037..c67a6364c 100644 --- a/crates/j2k-jpeg/src/entropy/progressive/allocation.rs +++ b/crates/j2k-jpeg/src/entropy/progressive/allocation.rs @@ -8,6 +8,7 @@ use crate::allocation::{ checked_add_allocation_bytes, checked_allocation_bytes, checked_allocation_len, try_reserve_for_len_with_live_budget, }; +use crate::color::scaled_sampling::ScaledSampling; use crate::error::JpegError; use super::model::{PreparedProgressiveComponentPlan, PreparedProgressivePlan}; @@ -129,11 +130,13 @@ pub(super) fn checked_phase_capacity( Ok(requested) } +/// Component images at each component's reduced IDCT size. pub(super) fn allocate_component_images( plan: &PreparedProgressivePlan, + scaled: &ScaledSampling, initial_live_bytes: usize, ) -> Result, JpegError> { - let planned_bytes = validate_component_image_workspace(&plan.components)?; + let planned_bytes = validate_component_image_workspace(&plan.components, scaled)?; checked_phase_capacity(initial_live_bytes, planned_bytes, plan.scratch_bytes)?; let mut live_bytes = initial_live_bytes; @@ -144,9 +147,8 @@ pub(super) fn allocate_component_images( &mut live_bytes, plan.scratch_bytes, )?; - for component in &plan.components { - let stride = checked_allocation_len::(component.block_cols as usize, 8)?; - let rows = checked_allocation_len::(component.block_rows as usize, 8)?; + for (index, component) in plan.components.iter().enumerate() { + let (stride, rows) = component_image_extent(component, scaled, index)?; let plane_len = checked_allocation_len::(stride, rows)?; let mut plane = Vec::new(); try_reserve_for_len_with_live_budget( @@ -161,13 +163,25 @@ pub(super) fn allocate_component_images( Ok(images) } +fn component_image_extent( + component: &PreparedProgressiveComponentPlan, + scaled: &ScaledSampling, + index: usize, +) -> Result<(usize, usize), JpegError> { + let idct_size = scaled.component(index).idct_size as usize; + Ok(( + checked_allocation_len::(component.block_cols as usize, idct_size)?, + checked_allocation_len::(component.block_rows as usize, idct_size)?, + )) +} + fn validate_component_image_workspace( components: &[PreparedProgressiveComponentPlan], + scaled: &ScaledSampling, ) -> Result { let mut total = checked_allocation_bytes::(components.len())?; - for component in components { - let stride = checked_allocation_len::(component.block_cols as usize, 8)?; - let rows = checked_allocation_len::(component.block_rows as usize, 8)?; + for (index, component) in components.iter().enumerate() { + let (stride, rows) = component_image_extent(component, scaled, index)?; let plane_len = checked_allocation_len::(stride, rows)?; total = checked_add_allocation_bytes(total, checked_allocation_bytes::(plane_len)?)?; } diff --git a/crates/j2k-jpeg/src/entropy/progressive/render.rs b/crates/j2k-jpeg/src/entropy/progressive/render.rs index 1b4d0eac7..5d843c2d4 100644 --- a/crates/j2k-jpeg/src/entropy/progressive/render.rs +++ b/crates/j2k-jpeg/src/entropy/progressive/render.rs @@ -6,19 +6,24 @@ use alloc::vec::Vec; use crate::allocation::{checked_allocation_len, try_reserve_for_len_with_live_budget}; use crate::backend::Backend; +use crate::color::scaled_sampling::{ScaledComponent, ScaledSampling, Upsample}; use crate::color::upsample::{upsample_h1v2_fancy_row, upsample_h2v1_fancy_row}; use crate::entropy::block::clamp_i16; use crate::entropy::ZIGZAG; use crate::error::JpegError; +use crate::idct::downscale; use crate::info::ColorSpace; use crate::output::OutputWriter; use super::allocation::{allocate_component_images, checked_phase_capacity, ComponentImage}; -use super::model::{PreparedProgressiveComponentPlan, PreparedProgressivePlan}; +use super::model::PreparedProgressivePlan; +/// IDCTs every block into its component image at the component's reduced +/// size (8, 4, 2 or 1 samples per block side, as libjpeg-turbo chooses). pub(super) fn render_component_images( plan: &PreparedProgressivePlan, backend: Backend, + scaled: &ScaledSampling, coeffs: &[Vec<[i32; 64]>], coefficient_live_bytes: usize, ) -> Result, JpegError> { @@ -27,22 +32,50 @@ pub(super) fn render_component_images( reason: "progressive coefficient/component count mismatch", }); } - let mut images = allocate_component_images(plan, coefficient_live_bytes)?; - for ((component, component_coeffs), image) in plan + let mut images = allocate_component_images(plan, scaled, coefficient_live_bytes)?; + for (((index, component), component_coeffs), image) in plan .components .iter() + .enumerate() .zip(coeffs.iter()) .zip(images.iter_mut()) { + let idct_size = scaled.component(index).idct_size as usize; let mut dequant = [0i16; 64]; let mut pixels = [0u8; 64]; + let mut pixels_4x4 = [0u8; 16]; + let mut pixels_2x2 = [0u8; 4]; let natural_quant = natural_order_quant(&component.quant); for by in 0..component.block_rows as usize { for bx in 0..component.block_cols as usize { let block_index = by * component.block_cols as usize + bx; dequantize_block(&component_coeffs[block_index], &natural_quant, &mut dequant); - backend.idct(&dequant, &mut pixels); - deposit_block(&mut image.plane, image.stride, bx * 8, by * 8, &pixels); + let block: &[u8] = match idct_size { + 8 => { + backend.idct(&dequant, &mut pixels); + &pixels + } + 4 => { + downscale::idct_islow_4x4(&dequant, &mut pixels_4x4); + &pixels_4x4 + } + 2 => { + downscale::idct_islow_2x2(&dequant, &mut pixels_2x2); + &pixels_2x2 + } + _ => { + pixels[0] = downscale::idct_islow_1x1(&dequant); + &pixels[..1] + } + }; + deposit_block( + &mut image.plane, + image.stride, + bx * idct_size, + by * idct_size, + idct_size, + block, + ); } } } @@ -67,22 +100,25 @@ fn dequantize_block(coeffs: &[i32; 64], natural_quant: &[u16; 64], out: &mut [i1 } } -fn deposit_block(plane: &mut [u8], stride: usize, x: usize, y: usize, block: &[u8; 64]) { - for row in 0..8 { +fn deposit_block(plane: &mut [u8], stride: usize, x: usize, y: usize, size: usize, block: &[u8]) { + for row in 0..size { let dst = (y + row) * stride + x; - let src = row * 8; - plane[dst..dst + 8].copy_from_slice(&block[src..src + 8]); + let src = row * size; + plane[dst..dst + size].copy_from_slice(&block[src..src + size]); } } +/// Upsamples the component images to the (scaled) output grid row by row. pub(super) fn emit_component_images( plan: &PreparedProgressivePlan, backend: Backend, + scaled: &ScaledSampling, + output_dimensions: (u32, u32), images: &[ComponentImage], image_live_bytes: usize, writer: &mut W, ) -> Result<(), JpegError> { - let (width, height) = plan.dimensions; + let (width, height) = output_dimensions; let width_usize = width as usize; if plan.components.len() == 1 { checked_phase_capacity(image_live_bytes, width_usize, plan.scratch_bytes)?; @@ -99,7 +135,7 @@ pub(super) fn emit_component_images( )?; gray.resize(width_usize, 0u8); for y in 0..height { - upsample_component_row(plan, backend, 0, image, y, &mut gray); + upsample_component_row(backend, scaled.component(0), image, y, &mut gray); writer.write_gray_row(y, &gray)?; } return Ok(()); @@ -121,9 +157,15 @@ pub(super) fn emit_component_images( try_reserve_for_len_with_live_budget(&mut c, width_usize, &mut live_bytes, plan.scratch_bytes)?; c.resize(width_usize, 0u8); for y in 0..height { - upsample_component_row(plan, backend, first, &images[first], y, &mut a); - upsample_component_row(plan, backend, second, &images[second], y, &mut b); - upsample_component_row(plan, backend, third, &images[third], y, &mut c); + upsample_component_row(backend, scaled.component(first), &images[first], y, &mut a); + upsample_component_row( + backend, + scaled.component(second), + &images[second], + y, + &mut b, + ); + upsample_component_row(backend, scaled.component(third), &images[third], y, &mut c); match plan.color_space { ColorSpace::YCbCr => writer.write_ycbcr_row(y, &a, &b, &c)?, ColorSpace::Rgb => writer.write_rgb_row(y, &a, &b, &c)?, @@ -148,69 +190,52 @@ fn component_slot(plan: &PreparedProgressivePlan, output_index: usize) -> Result }) } +/// Emits output row `y` of one component with libjpeg-turbo's upsampler for +/// it; rows past the component's last sample row replicate that row. fn upsample_component_row( - plan: &PreparedProgressivePlan, backend: Backend, - component_index: usize, + component: ScaledComponent, image: &ComponentImage, y: u32, out: &mut [u8], ) { - let component = &plan.components[component_index]; - let h_ratio = plan.sampling.max_h / component.h; - let v_ratio = plan.sampling.max_v / component.v; - if h_ratio == 1 && v_ratio == 1 { - let sample_y = (y as usize).min(component.sample_height.saturating_sub(1) as usize); - let row = component_row(component, image, sample_y); - out.copy_from_slice(&row[..out.len()]); - } else if h_ratio == 2 && v_ratio == 1 { - let sample_y = (y as usize).min(component.sample_height.saturating_sub(1) as usize); - let row = component_row(component, image, sample_y); - upsample_h2v1_fancy_row(row, out.len(), out); - } else if h_ratio == 2 && v_ratio == 2 { - let sample_y = ((y / 2) as usize).min(component.sample_height.saturating_sub(1) as usize); - let prev_y = sample_y.saturating_sub(1); - let next_y = (sample_y + 1).min(component.sample_height.saturating_sub(1) as usize); - let prev = component_row(component, image, prev_y); - let curr = component_row(component, image, sample_y); - let next = component_row(component, image, next_y); - backend.upsample_h2v2_fancy_row([prev, curr, next], out.len(), y % 2 == 1, out); - } else if h_ratio == 1 && v_ratio == 2 { - let sample_y = ((y / 2) as usize).min(component.sample_height.saturating_sub(1) as usize); - let prev_y = sample_y.saturating_sub(1); - let next_y = (sample_y + 1).min(component.sample_height.saturating_sub(1) as usize); - let prev = component_row(component, image, prev_y); - let curr = component_row(component, image, sample_y); - let next = component_row(component, image, next_y); - upsample_h1v2_fancy_row(prev, curr, next, out.len(), y % 2 == 1, out); - } else { - upsample_nearest(plan, component, image, y, out); + let last_row = component.height.saturating_sub(1) as usize; + let row = |sample_y: usize| component_row(component, image, sample_y.min(last_row)); + let y = y as usize; + match component.upsample { + Upsample::None => out.copy_from_slice(&row(y)[..out.len()]), + Upsample::FancyH2V1 => upsample_h2v1_fancy_row(row(y), out.len(), out), + Upsample::FancyH2V2 => { + let sample_y = y / 2; + let rows = [ + row(sample_y.saturating_sub(1)), + row(sample_y), + row(sample_y + 1), + ]; + backend.upsample_h2v2_fancy_row(rows, out.len(), y % 2 == 1, out); + } + Upsample::FancyH1V2 => { + let sample_y = y / 2; + upsample_h1v2_fancy_row( + row(sample_y.saturating_sub(1)), + row(sample_y), + row(sample_y + 1), + out.len(), + y % 2 == 1, + out, + ); + } + Upsample::Replicate => { + let samples = row(y / component.v_ratio as usize); + let h_ratio = component.h_ratio as usize; + for (x, dst) in out.iter_mut().enumerate() { + *dst = samples[(x / h_ratio).min(samples.len() - 1)]; + } + } } } -fn component_row<'a>( - component: &PreparedProgressiveComponentPlan, - image: &'a ComponentImage, - y: usize, -) -> &'a [u8] { - let width = component.sample_width as usize; +fn component_row(component: ScaledComponent, image: &ComponentImage, y: usize) -> &[u8] { let row_start = y * image.stride; - &image.plane[row_start..row_start + width] -} - -fn upsample_nearest( - plan: &PreparedProgressivePlan, - component: &PreparedProgressiveComponentPlan, - image: &ComponentImage, - y: u32, - out: &mut [u8], -) { - let sample_y = ((y as usize) * usize::from(component.v) / usize::from(plan.sampling.max_v)) - .min(component.sample_height.saturating_sub(1) as usize); - let row = component_row(component, image, sample_y); - for (x, dst) in out.iter_mut().enumerate() { - let sample_x = (x * usize::from(component.h) / usize::from(plan.sampling.max_h)) - .min(row.len().saturating_sub(1)); - *dst = row[sample_x]; - } + &image.plane[row_start..row_start + component.width as usize] } diff --git a/crates/j2k-jpeg/src/entropy/sequential.rs b/crates/j2k-jpeg/src/entropy/sequential.rs index b72fce4b6..f00a33e4f 100644 --- a/crates/j2k-jpeg/src/entropy/sequential.rs +++ b/crates/j2k-jpeg/src/entropy/sequential.rs @@ -21,13 +21,10 @@ pub(crate) use self::dct::{ }; #[cfg(feature = "bench-internals")] pub(crate) use self::fast420::decode_scan_fast_tile_rgb_profiled; -pub(crate) use self::fast420::{ - decode_scan_fast_tile_rgb, decode_scan_fast_tile_rgb_region, - decode_scan_fast_tile_rgb_region_scaled, FastTileRegionScaledRequest, -}; +pub(crate) use self::fast420::{decode_scan_fast_tile_rgb, decode_scan_fast_tile_rgb_region}; pub(crate) use self::generic::{decode_scan_baseline, decode_scan_baseline_rgb}; pub(crate) use self::layout::{fast_tile_region_first_decode_mcu, stripe_region_layout}; -use self::layout::{is_ycbcr_420, scaled_dimensions, Fast420RegionLayout}; +use self::layout::{scaled_dimensions, uses_fancy_420_emit, Fast420RegionLayout}; use self::output_scratch::{OutputScratch, RgbOutputScratch}; pub(crate) use self::plan::{ PreparedComponentPlan, PreparedDecodePlan, ResolvedPreparedComponentPlan, diff --git a/crates/j2k-jpeg/src/entropy/sequential/deposit.rs b/crates/j2k-jpeg/src/entropy/sequential/deposit.rs index 05e6b919b..a6db24c55 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/deposit.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/deposit.rs @@ -2,13 +2,7 @@ use super::{ResolvedPreparedComponentPlan, StripeBuffer}; use crate::backend::Backend; -use crate::entropy::block::{ - decode_block_for_1x1_idct, decode_block_for_reduced_idct, BlockActivity, CoefficientBlock, - ReducedIdctCoefficients, -}; -use crate::error::JpegError; -use crate::idct::downscale; -use crate::info::DownscaleFactor; +use crate::entropy::block::{BlockActivity, CoefficientBlock}; use crate::internal::bit_reader::BitReader; use core::ptr; @@ -132,17 +126,6 @@ pub(super) fn deposit_block_1x1(plane: &mut [u8], stride: usize, x: u32, y: u32, plane[dst] = pixel; } -pub(super) struct EntropyBlockState<'a, 'b> { - pub(super) br: &'a mut BitReader<'b>, - pub(super) prev_dc: &'a mut i32, - pub(super) coeff: &'a mut CoefficientBlock, -} - -pub(super) struct ReducedIdctScratch<'a> { - pub(super) pixels_4x4: &'a mut [u8; 16], - pub(super) pixels_2x2: &'a mut [u8; 4], -} - pub(super) struct PlaneBlockTarget<'a> { pub(super) plane: &'a mut [u8], pub(super) stride: usize, @@ -245,129 +228,3 @@ impl FastTile420Window { mx - self.stripe_mcu_start } } - -#[expect( - clippy::needless_pass_by_value, - reason = "entropy state, scratch, and plane target are compact borrowing descriptors consumed as one hot-path operation" -)] -pub(super) fn decode_scaled_block_to_plane( - comp: ResolvedPreparedComponentPlan<'_>, - downscale: DownscaleFactor, - state: EntropyBlockState<'_, '_>, - scratch: ReducedIdctScratch<'_>, - target: PlaneBlockTarget<'_>, -) -> Result<(), JpegError> { - let keep = match downscale { - DownscaleFactor::Full => unreachable!("scaled block path excludes full-size decode"), - DownscaleFactor::Half => ReducedIdctCoefficients::Half, - DownscaleFactor::Quarter => ReducedIdctCoefficients::Quarter, - DownscaleFactor::Eighth => { - decode_block_for_1x1_idct( - state.br, - comp.dc_table, - comp.ac_table, - state.prev_dc, - comp.quant, - state.coeff, - )?; - let pixel = downscale::idct_islow_1x1(state.coeff.coefficients()); - deposit_block_1x1(target.plane, target.stride, target.x, target.y, pixel); - return Ok(()); - } - }; - let dc_only = decode_block_for_reduced_idct( - state.br, - comp.dc_table, - comp.ac_table, - state.prev_dc, - comp.quant, - state.coeff, - keep, - )?; - match downscale { - DownscaleFactor::Full => unreachable!("scaled block path excludes full-size decode"), - DownscaleFactor::Half => { - if dc_only { - downscale::idct_islow_4x4_dc_only(state.coeff.dc_coeff(), scratch.pixels_4x4); - } else { - downscale::idct_islow_4x4(state.coeff.coefficients(), scratch.pixels_4x4); - } - deposit_block_4x4( - target.plane, - target.stride, - target.x, - target.y, - scratch.pixels_4x4, - ); - } - DownscaleFactor::Quarter => { - if dc_only { - downscale::idct_islow_2x2_dc_only(state.coeff.dc_coeff(), scratch.pixels_2x2); - } else { - downscale::idct_islow_2x2(state.coeff.coefficients(), scratch.pixels_2x2); - } - deposit_block_2x2( - target.plane, - target.stride, - target.x, - target.y, - *scratch.pixels_2x2, - ); - } - DownscaleFactor::Eighth => { - let pixel = downscale::idct_islow_1x1(state.coeff.coefficients()); - deposit_block_1x1(target.plane, target.stride, target.x, target.y, pixel); - } - } - Ok(()) -} - -#[expect( - clippy::needless_pass_by_value, - reason = "entropy state and plane target are compact borrowing descriptors used as one reduced-IDCT operation" -)] -pub(super) fn decode_quarter_block_to_plane( - comp: ResolvedPreparedComponentPlan<'_>, - pixels_2x2: &mut [u8; 4], - state: EntropyBlockState<'_, '_>, - target: PlaneBlockTarget<'_>, -) -> Result<(), JpegError> { - let dc_only = decode_block_for_reduced_idct( - state.br, - comp.dc_table, - comp.ac_table, - state.prev_dc, - comp.quant, - state.coeff, - ReducedIdctCoefficients::Quarter, - )?; - if dc_only { - downscale::idct_islow_2x2_dc_only(state.coeff.dc_coeff(), pixels_2x2); - } else { - downscale::idct_islow_2x2(state.coeff.coefficients(), pixels_2x2); - } - deposit_block_2x2(target.plane, target.stride, target.x, target.y, *pixels_2x2); - Ok(()) -} - -#[expect( - clippy::needless_pass_by_value, - reason = "entropy state and plane target are compact borrowing descriptors used as one reduced-IDCT operation" -)] -pub(super) fn decode_eighth_block_to_plane( - comp: ResolvedPreparedComponentPlan<'_>, - state: EntropyBlockState<'_, '_>, - target: PlaneBlockTarget<'_>, -) -> Result<(), JpegError> { - decode_block_for_1x1_idct( - state.br, - comp.dc_table, - comp.ac_table, - state.prev_dc, - comp.quant, - state.coeff, - )?; - let pixel = downscale::idct_islow_1x1(state.coeff.coefficients()); - deposit_block_1x1(target.plane, target.stride, target.x, target.y, pixel); - Ok(()) -} diff --git a/crates/j2k-jpeg/src/entropy/sequential/emit/four_component.rs b/crates/j2k-jpeg/src/entropy/sequential/emit/four_component.rs index 473af5bc1..0e454d452 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/emit/four_component.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/emit/four_component.rs @@ -9,7 +9,7 @@ use super::{ }; use crate::{ color::cmyk::{inverted_cmyk_to_rgb, ycck_to_rgb}, - error::JpegError, + color::scaled_sampling::ScaledSampling, info::ColorSpace, internal::scratch::RgbGenericRows, }; @@ -25,46 +25,22 @@ pub(super) struct FourComponentRow { pub(super) fn fill_four_component_rgb_row( plan: &PreparedDecodePlan, + scaled: &ScaledSampling, neighbors: StripeNeighbors<'_>, row: FourComponentRow, scratch: &mut RgbGenericRows, -) -> Result<(), JpegError> { +) { let FourComponentRow { local_y, stripe_rows, width, } = row; - let (c0_h, c0_v) = plan - .sampling - .component(0) - .map(|(h, v)| (u32::from(h), u32::from(v))) - .ok_or(JpegError::UnsupportedComponentCount { count: 0 })?; - let (c1_h, c1_v) = plan - .sampling - .component(1) - .map(|(h, v)| (u32::from(h), u32::from(v))) - .ok_or(JpegError::UnsupportedComponentCount { count: 1 })?; - let (c2_h, c2_v) = plan - .sampling - .component(2) - .map(|(h, v)| (u32::from(h), u32::from(v))) - .ok_or(JpegError::UnsupportedComponentCount { count: 2 })?; - let (k_h, k_v) = plan - .sampling - .component(3) - .map(|(h, v)| (u32::from(h), u32::from(v))) - .ok_or(JpegError::UnsupportedComponentCount { count: 3 })?; - let max_h = u32::from(plan.sampling.max_h); - let max_v = u32::from(plan.sampling.max_v); upsample_component_row_stripe(StripeComponentUpsample { neighbors, spec: StripeComponentUpsampleSpec { plane_idx: 0, - comp_h: c0_h, - comp_v: c0_v, - max_h, - max_v, + component: scaled.component(0), local_y_out: local_y, stripe_rows, width, @@ -75,10 +51,7 @@ pub(super) fn fill_four_component_rgb_row( neighbors, spec: StripeComponentUpsampleSpec { plane_idx: 1, - comp_h: c1_h, - comp_v: c1_v, - max_h, - max_v, + component: scaled.component(1), local_y_out: local_y, stripe_rows, width, @@ -89,10 +62,7 @@ pub(super) fn fill_four_component_rgb_row( neighbors, spec: StripeComponentUpsampleSpec { plane_idx: 2, - comp_h: c2_h, - comp_v: c2_v, - max_h, - max_v, + component: scaled.component(2), local_y_out: local_y, stripe_rows, width, @@ -103,10 +73,7 @@ pub(super) fn fill_four_component_rgb_row( neighbors, spec: StripeComponentUpsampleSpec { plane_idx: 3, - comp_h: k_h, - comp_v: k_v, - max_h, - max_v, + component: scaled.component(3), local_y_out: local_y, stripe_rows, width, @@ -126,6 +93,4 @@ pub(super) fn fill_four_component_rgb_row( scratch.g[x] = g; scratch.b[x] = b; } - - Ok(()) } diff --git a/crates/j2k-jpeg/src/entropy/sequential/emit/output.rs b/crates/j2k-jpeg/src/entropy/sequential/emit/output.rs index 14191dd21..32b795a4f 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/emit/output.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/emit/output.rs @@ -1,9 +1,9 @@ // SPDX-License-Identifier: MIT OR Apache-2.0 use super::{ - super::{is_ycbcr_420, scaled_dimensions, OutputScratch, PreparedDecodePlan}, + super::{scaled_dimensions, uses_fancy_420_emit, OutputScratch, PreparedDecodePlan}, four_component::{fill_four_component_rgb_row, FourComponentRow}, - types::{StripeEmit, StripeNeighbors}, + types::{ensure_color_components, StripeEmit, StripeNeighbors}, upsample::{ upsample_420_pair, upsample_component_row_stripe, Stripe420PairSpec, Stripe420PairUpsample, StripeComponentUpsample, StripeComponentUpsampleSpec, @@ -32,6 +32,7 @@ pub(in crate::entropy::sequential) fn emit_stripe( stripe_index, source_width, downscale, + scaled, } = emit; let max_v = u32::from(plan.sampling.max_v); let mcu_height_px = downscale.output_block_size() * max_v; @@ -43,6 +44,7 @@ pub(in crate::entropy::sequential) fn emit_stripe( if stripe_rows == 0 { return Ok(()); } + ensure_color_components(plan)?; let width = source_width; let neighbors = StripeNeighbors { prev, curr, next }; @@ -54,27 +56,10 @@ pub(in crate::entropy::sequential) fn emit_stripe( } } ColorSpace::YCbCr => { - let (cb_h, cb_v) = plan - .sampling - .component(1) - .map(|(h, v)| (u32::from(h), u32::from(v))) - .ok_or(JpegError::UnsupportedComponentCount { count: 1 })?; - let (cr_h, cr_v) = plan - .sampling - .component(2) - .map(|(h, v)| (u32::from(h), u32::from(v))) - .ok_or(JpegError::UnsupportedComponentCount { count: 2 })?; - - let max_h = u32::from(plan.sampling.max_h); - let max_v = u32::from(plan.sampling.max_v); - - if is_ycbcr_420(plan) { + if uses_fancy_420_emit(plan, scaled) { let OutputScratch::YCbCr420(scratch) = output_scratch else { unreachable!("4:2:0 YCbCr requires dedicated scratch"); }; - debug_assert!( - max_h == 2 && max_v == 2 && cb_h == 1 && cb_v == 1 && cr_h == 1 && cr_v == 1 - ); let mut local_y = 0usize; while local_y < stripe_rows { @@ -133,10 +118,7 @@ pub(in crate::entropy::sequential) fn emit_stripe( neighbors, spec: StripeComponentUpsampleSpec { plane_idx: 1, - comp_h: cb_h, - comp_v: cb_v, - max_h, - max_v, + component: scaled.component(1), local_y_out: local_y as u32, stripe_rows, width, @@ -147,10 +129,7 @@ pub(in crate::entropy::sequential) fn emit_stripe( neighbors, spec: StripeComponentUpsampleSpec { plane_idx: 2, - comp_h: cr_h, - comp_v: cr_v, - max_h, - max_v, + component: scaled.component(2), local_y_out: local_y as u32, stripe_rows, width, @@ -167,25 +146,6 @@ pub(in crate::entropy::sequential) fn emit_stripe( } } ColorSpace::Rgb => { - let (r_h, r_v) = plan - .sampling - .component(0) - .map(|(h, v)| (u32::from(h), u32::from(v))) - .ok_or(JpegError::UnsupportedComponentCount { count: 0 })?; - let (g_h, g_v) = plan - .sampling - .component(1) - .map(|(h, v)| (u32::from(h), u32::from(v))) - .ok_or(JpegError::UnsupportedComponentCount { count: 1 })?; - let (b_h, b_v) = plan - .sampling - .component(2) - .map(|(h, v)| (u32::from(h), u32::from(v))) - .ok_or(JpegError::UnsupportedComponentCount { count: 2 })?; - - let max_h = u32::from(plan.sampling.max_h); - let max_v = u32::from(plan.sampling.max_v); - let OutputScratch::RgbGeneric(scratch) = output_scratch else { unreachable!("RGB decode requires reusable row scratch"); }; @@ -195,10 +155,7 @@ pub(in crate::entropy::sequential) fn emit_stripe( neighbors, spec: StripeComponentUpsampleSpec { plane_idx: 0, - comp_h: r_h, - comp_v: r_v, - max_h, - max_v, + component: scaled.component(0), local_y_out: local_y as u32, stripe_rows, width, @@ -209,10 +166,7 @@ pub(in crate::entropy::sequential) fn emit_stripe( neighbors, spec: StripeComponentUpsampleSpec { plane_idx: 1, - comp_h: g_h, - comp_v: g_v, - max_h, - max_v, + component: scaled.component(1), local_y_out: local_y as u32, stripe_rows, width, @@ -223,10 +177,7 @@ pub(in crate::entropy::sequential) fn emit_stripe( neighbors, spec: StripeComponentUpsampleSpec { plane_idx: 2, - comp_h: b_h, - comp_v: b_v, - max_h, - max_v, + component: scaled.component(2), local_y_out: local_y as u32, stripe_rows, width, @@ -248,6 +199,7 @@ pub(in crate::entropy::sequential) fn emit_stripe( for local_y in 0..stripe_rows { fill_four_component_rgb_row( plan, + scaled, neighbors, FourComponentRow { local_y: local_y as u32, @@ -255,7 +207,7 @@ pub(in crate::entropy::sequential) fn emit_stripe( width, }, scratch, - )?; + ); writer.write_rgb_row( y_start + local_y as u32, &scratch.r[..width], diff --git a/crates/j2k-jpeg/src/entropy/sequential/emit/rgb.rs b/crates/j2k-jpeg/src/entropy/sequential/emit/rgb.rs index ae0898315..d465250c6 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/emit/rgb.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/emit/rgb.rs @@ -1,9 +1,9 @@ // SPDX-License-Identifier: MIT OR Apache-2.0 use super::{ - super::{is_ycbcr_420, scaled_dimensions, PreparedDecodePlan, RgbOutputScratch}, + super::{scaled_dimensions, uses_fancy_420_emit, PreparedDecodePlan, RgbOutputScratch}, four_component::{fill_four_component_rgb_row, FourComponentRow}, - types::{StripeEmit, StripeNeighbors}, + types::{ensure_color_components, StripeEmit, StripeNeighbors}, upsample::{ component_row_triplet, upsample_component_row_stripe, valid_component_rows, StripeComponentUpsample, StripeComponentUpsampleSpec, @@ -38,6 +38,7 @@ pub(in crate::entropy::sequential) fn emit_stripe_rgb { + ColorSpace::YCbCr if uses_fancy_420_emit(plan, scaled) => { let RgbOutputScratch::YCbCr420 = output_scratch else { unreachable!("4:2:0 YCbCr RGB output requires dedicated scratch"); }; @@ -119,21 +121,10 @@ pub(in crate::entropy::sequential) fn emit_stripe_rgb { pub(in crate::entropy::sequential) stripe_index: u32, pub(in crate::entropy::sequential) source_width: usize, pub(in crate::entropy::sequential) downscale: DownscaleFactor, + pub(in crate::entropy::sequential) scaled: &'a ScaledSampling, } #[derive(Clone, Copy)] @@ -31,3 +34,21 @@ pub(in crate::entropy::sequential) struct StripeNeighbors<'a> { pub(in crate::entropy::sequential) curr: &'a StripeBuffer, pub(in crate::entropy::sequential) next: Option<&'a StripeBuffer>, } + +/// Rejects a plan with fewer components than its color space's emitters read +/// (three for YCbCr and RGB, four for CMYK and YCCK). +pub(in crate::entropy::sequential) fn ensure_color_components( + plan: &PreparedDecodePlan, +) -> Result<(), JpegError> { + let needed = match plan.color_space { + ColorSpace::Grayscale => 1, + ColorSpace::YCbCr | ColorSpace::Rgb => 3, + ColorSpace::Cmyk | ColorSpace::Ycck => 4, + }; + if plan.sampling.len() < needed { + return Err(JpegError::UnsupportedComponentCount { + count: u8::try_from(plan.sampling.len()).unwrap_or(u8::MAX), + }); + } + Ok(()) +} diff --git a/crates/j2k-jpeg/src/entropy/sequential/emit/upsample.rs b/crates/j2k-jpeg/src/entropy/sequential/emit/upsample.rs index ce0b545b0..303d57c84 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/emit/upsample.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/emit/upsample.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: MIT OR Apache-2.0 use super::{super::StripePlane, types::StripeNeighbors}; +use crate::color::scaled_sampling::{ScaledComponent, Upsample}; use crate::color::upsample::{ upsample_1x1, upsample_h1v2_fancy_row, upsample_h2v1_fancy_row, upsample_h2v2_fancy_row, upsample_h2v2_fancy_rows, @@ -9,10 +10,7 @@ use crate::color::upsample::{ #[derive(Clone, Copy)] pub(super) struct StripeComponentUpsampleSpec { pub(super) plane_idx: usize, - pub(super) comp_h: u32, - pub(super) comp_v: u32, - pub(super) max_h: u32, - pub(super) max_v: u32, + pub(super) component: ScaledComponent, pub(super) local_y_out: u32, pub(super) stripe_rows: usize, pub(super) width: usize, @@ -106,16 +104,17 @@ pub(super) fn upsample_component_row_stripe(request: StripeComponentUpsample<'_, let StripeNeighbors { prev, curr, next } = neighbors; let StripeComponentUpsampleSpec { plane_idx, - comp_h, - comp_v, - max_h, - max_v, + component, local_y_out, stripe_rows, width, } = spec; - let v_ratio = max_v / comp_v; - let h_ratio = max_h / comp_h; + let ScaledComponent { + h_ratio, + v_ratio, + upsample, + .. + } = component; let curr_plane = curr.plane(plane_idx); let chroma_rows = curr_plane.rows as u32; let chroma_y = (local_y_out / v_ratio).min(chroma_rows.saturating_sub(1)); @@ -127,15 +126,15 @@ pub(super) fn upsample_component_row_stripe(request: StripeComponentUpsample<'_, valid_component_rows(stripe_rows, v_ratio as usize), ); - match (h_ratio, v_ratio) { - (1, 1) => { + match upsample { + Upsample::None => { upsample_1x1(&curr_row[..width], out); } - (2, 1) => { + Upsample::FancyH2V1 => { let chroma_cols = width.div_ceil(2); upsample_h2v1_fancy_row(&curr_row[..chroma_cols], width, out); } - (1, 2) => { + Upsample::FancyH1V2 => { upsample_h1v2_fancy_row( &prev_row[..width], &curr_row[..width], @@ -145,7 +144,7 @@ pub(super) fn upsample_component_row_stripe(request: StripeComponentUpsample<'_, out, ); } - (2, 2) => { + Upsample::FancyH2V2 => { let chroma_cols = width.div_ceil(2); upsample_h2v2_fancy_row( &prev_row[..chroma_cols], @@ -156,7 +155,7 @@ pub(super) fn upsample_component_row_stripe(request: StripeComponentUpsample<'_, out, ); } - _ => { + Upsample::Replicate => { for (x, slot) in out.iter_mut().enumerate().take(width) { let cx = ((x as u32) / h_ratio).min(curr_row.len() as u32 - 1); *slot = curr_row[cx as usize]; diff --git a/crates/j2k-jpeg/src/entropy/sequential/fast420/mod.rs b/crates/j2k-jpeg/src/entropy/sequential/fast420/mod.rs index 30a77cee1..c453ad314 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/fast420/mod.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/fast420/mod.rs @@ -1,17 +1,17 @@ // SPDX-License-Identifier: MIT OR Apache-2.0 -//! Fast 4:2:0 sequential scan drivers, ROI planning, and scaled routing. +//! Fast 4:2:0 sequential scan drivers and ROI planning (full-size only; scaled +//! 4:2:0 decodes chroma with a larger IDCT on the generic route). use super::deposit::{ FastTile420Components, FastTile420DcState, FastTile420EntropyState, FastTile420Window, - ReducedIdctScratch, }; use super::emit::{ emit_stripe_rgb, emit_stripe_rgb_420_region, Fast420RegionStripe, StripeEmit, StripeNeighbors, }; use super::layout::{ fast420_decode_mcu_row_end, fast420_first_decode_mcu_row, last_mcu_row_for_rect, - mcu_row_intersects_rect, scaled_dimensions, Fast420RegionLayout, + mcu_row_intersects_rect, uses_fancy_420_emit, Fast420RegionLayout, }; use super::profile::{Fast420ScanProfiler, NoopFast420Profiler, NoopFast420ScanProfile}; use super::restart::{finish_scan, reader_from_checkpoint}; @@ -19,6 +19,7 @@ use super::{PreparedDecodePlan, ResolvedPreparedComponentPlan, RgbOutputScratch, use crate::backend::Backend; #[cfg(feature = "bench-internals")] use crate::bench_support::BenchFast420Profile; +use crate::color::scaled_sampling::ScaledSampling; use crate::entropy::block::CoefficientBlock; use crate::error::{JpegError, Warning}; use crate::info::{DownscaleFactor, Rect}; @@ -31,7 +32,7 @@ use alloc::vec::Vec; mod rows; pub(super) use self::rows::decode_mcu_row_fast_tile_420; -use self::rows::{decode_mcu_row_fast_tile_420_scaled, skip_mcu_fast_tile_420}; +use self::rows::skip_mcu_fast_tile_420; fn fast_tile_components( plan: &PreparedDecodePlan, @@ -90,10 +91,13 @@ where pool.prepare_for( plan, + plan.sampling, mcus_per_row, DownscaleFactor::Full.output_block_size(), plan.scratch_bytes, )?; + let scaled = ScaledSampling::new(plan.sampling, DownscaleFactor::Full, plan.dimensions); + debug_assert!(uses_fancy_420_emit(plan, &scaled)); let mut br = BitReader::new(scan_bytes); let mut coeff = CoefficientBlock::default(); @@ -179,6 +183,7 @@ where stripe_index: my - 1, source_width: width as usize, downscale: DownscaleFactor::Full, + scaled: &scaled, }, )?; profile.finish_rgb_emit(emit_timer); @@ -200,6 +205,7 @@ where stripe_index: mcu_rows - 1, source_width: width as usize, downscale: DownscaleFactor::Full, + scaled: &scaled, }, )?; profile.finish_rgb_emit(emit_timer); @@ -256,6 +262,7 @@ pub(crate) fn decode_scan_fast_tile_rgb_region { - pub(crate) roi: Rect, - pub(crate) downscale: DownscaleFactor, - pub(crate) checkpoint: Option<&'a DeviceCheckpoint>, -} - -#[expect( - clippy::too_many_lines, - reason = "the fused scaled-region loop keeps ROI seek, reduced IDCT, restart, and writer state in decode order" -)] -pub(crate) fn decode_scan_fast_tile_rgb_region_scaled( - plan: &PreparedDecodePlan, - backend: Backend, - scan_bytes: &[u8], - pool: &mut ScratchPool, - writer: &mut W, - request: FastTileRegionScaledRequest<'_>, -) -> Result, JpegError> { - let FastTileRegionScaledRequest { - roi, - downscale, - checkpoint, - } = request; - debug_assert!(plan.matches_fast_tile_shape()); - debug_assert!(downscale != DownscaleFactor::Full); - - let (width, height) = scaled_dimensions(plan.dimensions, downscale); - let max_h = u32::from(plan.sampling.max_h); - let max_v = u32::from(plan.sampling.max_v); - let block_size = downscale.output_block_size(); - let mcu_width_px = block_size * max_h; - let mcu_height_px = block_size * max_v; - let mcus_per_row = width.div_ceil(mcu_width_px); - let mcu_rows = height.div_ceil(mcu_height_px); - let first_decode_mcu_row = fast420_first_decode_mcu_row(roi, mcu_height_px); - let decode_mcu_row_end = fast420_decode_mcu_row_end(roi, mcu_height_px, mcu_rows); - let last_output_mcu_row = last_mcu_row_for_rect(roi, mcu_height_px, mcu_rows); - - let region_layout = Fast420RegionLayout::new_for_mcu_width(width as usize, roi, mcu_width_px); - - let mut crop_rows = pool.take_sink_rows( - region_layout.row_width().saturating_mul(3), - plan.scratch_bytes, - )?; - if let Err(error) = pool.prepare_for( - plan, - region_layout.stripe_mcus_per_row, - block_size, - plan.scratch_bytes, - ) { - pool.restore_sink_rows(crop_rows); - return Err(error); - } - let result = (|| { - let mut coeff = CoefficientBlock::default(); - let ScratchPool { - prev_dc, - stripe_a, - stripe_b, - stripe_c, - .. - } = pool; - let mut pixels_4x4 = [0u8; 16]; - let mut pixels_2x2 = [0u8; 4]; - let (y_dc_slice, rest_dc) = prev_dc.split_at_mut(1); - let (cb_dc_slice, cr_dc_slice) = rest_dc.split_at_mut(1); - let y_dc = &mut y_dc_slice[0]; - let cb_dc = &mut cb_dc_slice[0]; - let cr_dc = &mut cr_dc_slice[0]; - let target_mcu = first_decode_mcu_row * mcus_per_row; - let (mut br, checkpoint_dc, start_mcu) = - reader_from_checkpoint(scan_bytes, checkpoint, target_mcu); - *y_dc = checkpoint_dc[0]; - *cb_dc = checkpoint_dc[1]; - *cr_dc = checkpoint_dc[2]; - let (y, cb, cr) = fast_tile_components(plan)?; - let components = FastTile420Components { y, cb, cr }; - let window = FastTile420Window { - mcus_per_row, - stripe_mcu_start: region_layout.stripe_mcu_start, - stripe_mcus_per_row: region_layout.stripe_mcus_per_row, - }; - for _ in start_mcu..target_mcu { - skip_mcu_fast_tile_420( - components.y, - components.cb, - components.cr, - &mut br, - &mut *y_dc, - &mut *cb_dc, - &mut *cr_dc, - )?; - } - - let mut prev_stripe: &mut StripeBuffer = stripe_a; - let mut curr_stripe: &mut StripeBuffer = stripe_b; - let mut next_stripe: &mut StripeBuffer = stripe_c; - - decode_mcu_row_fast_tile_420_scaled( - components, - &mut FastTile420EntropyState { - br: &mut br, - dc: FastTile420DcState { - y: &mut *y_dc, - cb: &mut *cb_dc, - cr: &mut *cr_dc, - }, - coeff: &mut coeff, - }, - downscale, - ReducedIdctScratch { - pixels_4x4: &mut pixels_4x4, - pixels_2x2: &mut pixels_2x2, - }, - window, - curr_stripe, - )?; - - let mut has_prev = false; - for my in first_decode_mcu_row + 1..decode_mcu_row_end { - decode_mcu_row_fast_tile_420_scaled( - components, - &mut FastTile420EntropyState { - br: &mut br, - dc: FastTile420DcState { - y: &mut *y_dc, - cb: &mut *cb_dc, - cr: &mut *cr_dc, - }, - coeff: &mut coeff, - }, - downscale, - ReducedIdctScratch { - pixels_4x4: &mut pixels_4x4, - pixels_2x2: &mut pixels_2x2, - }, - window, - next_stripe, - )?; - if mcu_row_intersects_rect(my - 1, mcu_height_px, roi) { - emit_stripe_rgb_420_region( - plan, - backend, - writer, - Fast420RegionStripe { - neighbors: StripeNeighbors { - prev: has_prev.then_some(&*prev_stripe), - curr: curr_stripe, - next: Some(&*next_stripe), - }, - stripe_index: my - 1, - roi, - region_layout, - crop_rows: &mut crop_rows, - downscale, - }, - )?; - } - core::mem::swap(&mut prev_stripe, &mut curr_stripe); - core::mem::swap(&mut curr_stripe, &mut next_stripe); - has_prev = true; - } - - let curr_mcu_row = decode_mcu_row_end - 1; - if curr_mcu_row <= last_output_mcu_row - && mcu_row_intersects_rect(curr_mcu_row, mcu_height_px, roi) - { - emit_stripe_rgb_420_region( - plan, - backend, - writer, - Fast420RegionStripe { - neighbors: StripeNeighbors { - prev: has_prev.then_some(&*prev_stripe), - curr: curr_stripe, - next: None, - }, - stripe_index: curr_mcu_row, - roi, - region_layout, - crop_rows: &mut crop_rows, - downscale, - }, - )?; - } - finish_scan(&mut br, decode_mcu_row_end == mcu_rows) - })(); - pool.restore_sink_rows(crop_rows); - result -} diff --git a/crates/j2k-jpeg/src/entropy/sequential/fast420/rows.rs b/crates/j2k-jpeg/src/entropy/sequential/fast420/rows.rs index 5a7e5d909..17f5c4c88 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/fast420/rows.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/fast420/rows.rs @@ -3,17 +3,14 @@ //! Entropy and IDCT row kernels for the fast 4:2:0 sequential route. use super::super::deposit::{ - assert_stripe_deposit_capacity, decode_eighth_block_to_plane, decode_quarter_block_to_plane, - decode_scaled_block_to_plane, idct_deposit_fast_tile_block, EntropyBlockState, - FastTile420Components, FastTile420EntropyState, FastTile420Window, PlaneBlockTarget, - ReducedIdctScratch, + assert_stripe_deposit_capacity, idct_deposit_fast_tile_block, FastTile420Components, + FastTile420EntropyState, FastTile420Window, PlaneBlockTarget, }; use super::super::profile::Fast420Profiler; use super::super::{ResolvedPreparedComponentPlan, StripeBuffer}; use crate::backend::Backend; use crate::entropy::block::{decode_block_with_activity, skip_block}; use crate::error::JpegError; -use crate::info::DownscaleFactor; use crate::internal::bit_reader::BitReader; // ROI seeks invoke this leaf once per skipped MCU; forced inlining keeps its six Huffman skips @@ -40,175 +37,6 @@ pub(super) fn skip_mcu_fast_tile_420( Ok(()) } -#[expect( - clippy::too_many_lines, - reason = "the scaled MCU-row kernel keeps six component blocks and shared reduced-IDCT scratch in sampling order" -)] -#[expect( - clippy::needless_pass_by_value, - reason = "scratch and window are compact borrowing descriptors threaded through the fused 4:2:0 MCU-row hot path" -)] -pub(super) fn decode_mcu_row_fast_tile_420_scaled( - components: FastTile420Components<'_>, - state: &mut FastTile420EntropyState<'_, '_>, - downscale: DownscaleFactor, - scratch: ReducedIdctScratch<'_>, - window: FastTile420Window, - stripe: &mut StripeBuffer, -) -> Result<(), JpegError> { - if downscale == DownscaleFactor::Quarter { - return decode_mcu_row_fast_tile_420_quarter( - components, - state, - scratch.pixels_2x2, - window, - stripe, - ); - } - if downscale == DownscaleFactor::Eighth { - return decode_mcu_row_fast_tile_420_eighth(components, state, window, stripe); - } - - let block_size = downscale.output_block_size(); - let y_stride = stripe.plane_strides[0]; - let cb_stride = stripe.plane_strides[1]; - let cr_stride = stripe.plane_strides[2]; - - for mx in 0..window.mcus_per_row { - if !window.contains_mcu(mx) { - skip_mcu_fast_tile_420( - components.y, - components.cb, - components.cr, - &mut *state.br, - &mut *state.dc.y, - &mut *state.dc.cb, - &mut *state.dc.cr, - )?; - continue; - } - - let local_mx = window.local_mcu_x(mx); - let y_x = local_mx * 2 * block_size; - let c_x = local_mx * block_size; - decode_scaled_block_to_plane( - components.y, - downscale, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - ReducedIdctScratch { - pixels_4x4: &mut *scratch.pixels_4x4, - pixels_2x2: &mut *scratch.pixels_2x2, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x, - y: 0, - }, - )?; - decode_scaled_block_to_plane( - components.y, - downscale, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - ReducedIdctScratch { - pixels_4x4: &mut *scratch.pixels_4x4, - pixels_2x2: &mut *scratch.pixels_2x2, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x + block_size, - y: 0, - }, - )?; - decode_scaled_block_to_plane( - components.y, - downscale, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - ReducedIdctScratch { - pixels_4x4: &mut *scratch.pixels_4x4, - pixels_2x2: &mut *scratch.pixels_2x2, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x, - y: block_size, - }, - )?; - decode_scaled_block_to_plane( - components.y, - downscale, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - ReducedIdctScratch { - pixels_4x4: &mut *scratch.pixels_4x4, - pixels_2x2: &mut *scratch.pixels_2x2, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x + block_size, - y: block_size, - }, - )?; - decode_scaled_block_to_plane( - components.cb, - downscale, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.cb, - coeff: &mut *state.coeff, - }, - ReducedIdctScratch { - pixels_4x4: &mut *scratch.pixels_4x4, - pixels_2x2: &mut *scratch.pixels_2x2, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[1], - stride: cb_stride, - x: c_x, - y: 0, - }, - )?; - decode_scaled_block_to_plane( - components.cr, - downscale, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.cr, - coeff: &mut *state.coeff, - }, - ReducedIdctScratch { - pixels_4x4: &mut *scratch.pixels_4x4, - pixels_2x2: &mut *scratch.pixels_2x2, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[2], - stride: cr_stride, - x: c_x, - y: 0, - }, - )?; - } - Ok(()) -} - pub(in crate::entropy::sequential) fn decode_mcu_row_fast_tile_420( components: FastTile420Components<'_>, backend: Backend, @@ -417,252 +245,3 @@ fn decode_mcu_row_fast_tile_420_local( Ok(()) } - -#[expect( - clippy::too_many_lines, - reason = "the eighth-scale MCU kernel keeps six entropy blocks and 1x1 deposits in sampling order" -)] -fn decode_mcu_row_fast_tile_420_eighth( - components: FastTile420Components<'_>, - state: &mut FastTile420EntropyState<'_, '_>, - window: FastTile420Window, - stripe: &mut StripeBuffer, -) -> Result<(), JpegError> { - const BLOCK_SIZE: u32 = 1; - let y_stride = stripe.plane_strides[0]; - let cb_stride = stripe.plane_strides[1]; - let cr_stride = stripe.plane_strides[2]; - - for mx in 0..window.mcus_per_row { - if !window.contains_mcu(mx) { - skip_mcu_fast_tile_420( - components.y, - components.cb, - components.cr, - &mut *state.br, - &mut *state.dc.y, - &mut *state.dc.cb, - &mut *state.dc.cr, - )?; - continue; - } - - let local_mx = window.local_mcu_x(mx); - let y_x = local_mx * 2 * BLOCK_SIZE; - let c_x = local_mx * BLOCK_SIZE; - decode_eighth_block_to_plane( - components.y, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x, - y: 0, - }, - )?; - decode_eighth_block_to_plane( - components.y, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x + BLOCK_SIZE, - y: 0, - }, - )?; - decode_eighth_block_to_plane( - components.y, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x, - y: BLOCK_SIZE, - }, - )?; - decode_eighth_block_to_plane( - components.y, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x + BLOCK_SIZE, - y: BLOCK_SIZE, - }, - )?; - decode_eighth_block_to_plane( - components.cb, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.cb, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[1], - stride: cb_stride, - x: c_x, - y: 0, - }, - )?; - decode_eighth_block_to_plane( - components.cr, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.cr, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[2], - stride: cr_stride, - x: c_x, - y: 0, - }, - )?; - } - - Ok(()) -} - -#[expect( - clippy::too_many_lines, - reason = "the quarter-scale MCU kernel keeps six entropy blocks and 2x2 deposits in sampling order" -)] -fn decode_mcu_row_fast_tile_420_quarter( - components: FastTile420Components<'_>, - state: &mut FastTile420EntropyState<'_, '_>, - pixels_2x2: &mut [u8; 4], - window: FastTile420Window, - stripe: &mut StripeBuffer, -) -> Result<(), JpegError> { - const BLOCK_SIZE: u32 = 2; - let y_stride = stripe.plane_strides[0]; - let cb_stride = stripe.plane_strides[1]; - let cr_stride = stripe.plane_strides[2]; - - for mx in 0..window.mcus_per_row { - if !window.contains_mcu(mx) { - skip_mcu_fast_tile_420( - components.y, - components.cb, - components.cr, - &mut *state.br, - &mut *state.dc.y, - &mut *state.dc.cb, - &mut *state.dc.cr, - )?; - continue; - } - - let local_mx = window.local_mcu_x(mx); - let y_x = local_mx * 2 * BLOCK_SIZE; - let c_x = local_mx * BLOCK_SIZE; - decode_quarter_block_to_plane( - components.y, - pixels_2x2, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x, - y: 0, - }, - )?; - decode_quarter_block_to_plane( - components.y, - pixels_2x2, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x + BLOCK_SIZE, - y: 0, - }, - )?; - decode_quarter_block_to_plane( - components.y, - pixels_2x2, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x, - y: BLOCK_SIZE, - }, - )?; - decode_quarter_block_to_plane( - components.y, - pixels_2x2, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.y, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[0], - stride: y_stride, - x: y_x + BLOCK_SIZE, - y: BLOCK_SIZE, - }, - )?; - decode_quarter_block_to_plane( - components.cb, - pixels_2x2, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.cb, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[1], - stride: cb_stride, - x: c_x, - y: 0, - }, - )?; - decode_quarter_block_to_plane( - components.cr, - pixels_2x2, - EntropyBlockState { - br: &mut *state.br, - prev_dc: &mut *state.dc.cr, - coeff: &mut *state.coeff, - }, - PlaneBlockTarget { - plane: &mut stripe.planes[2], - stride: cr_stride, - x: c_x, - y: 0, - }, - )?; - } - - Ok(()) -} diff --git a/crates/j2k-jpeg/src/entropy/sequential/generic.rs b/crates/j2k-jpeg/src/entropy/sequential/generic.rs index 219760924..d1024b6eb 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/generic.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/generic.rs @@ -7,6 +7,7 @@ mod row; use self::driver::{decode_scan_rows, ScanBuffers, ScanOutputMode, ScanSetup, StripeEmitter}; use super::emit::{emit_stripe, emit_stripe_rgb, StripeEmit}; +use super::layout::uses_fancy_420_emit; use super::{OutputScratch, PreparedDecodePlan, RgbOutputScratch}; use crate::backend::Backend; use crate::error::{JpegError, Warning}; @@ -59,6 +60,7 @@ pub(crate) fn decode_scan_baseline( ) -> Result, JpegError> { let setup = ScanSetup::new(plan, downscale, output_rect, ScanOutputMode::ComponentRows); setup.prepare_pool(plan, pool)?; + let fancy_420 = uses_fancy_420_emit(plan, &setup.scaled); let ScratchPool { prev_dc, stripe_a, @@ -71,7 +73,7 @@ pub(crate) fn decode_scan_baseline( } = pool; let scratch = match plan.color_space { ColorSpace::Grayscale => OutputScratch::Grayscale, - ColorSpace::YCbCr if super::is_ycbcr_420(plan) => OutputScratch::YCbCr420(ycbcr_420_rows), + ColorSpace::YCbCr if fancy_420 => OutputScratch::YCbCr420(ycbcr_420_rows), ColorSpace::YCbCr => OutputScratch::YCbCrGeneric(ycbcr_generic_rows), ColorSpace::Rgb | ColorSpace::Cmyk | ColorSpace::Ycck => { OutputScratch::RgbGeneric(rgb_generic_rows) @@ -109,6 +111,7 @@ pub(crate) fn decode_scan_baseline_rgb( ) -> Result, JpegError> { let setup = ScanSetup::new(plan, downscale, output_rect, ScanOutputMode::InterleavedRgb); setup.prepare_pool(plan, pool)?; + let fancy_420 = uses_fancy_420_emit(plan, &setup.scaled); let ScratchPool { prev_dc, stripe_a, @@ -120,7 +123,7 @@ pub(crate) fn decode_scan_baseline_rgb( } = pool; let scratch = match plan.color_space { ColorSpace::Grayscale => RgbOutputScratch::None, - ColorSpace::YCbCr if super::is_ycbcr_420(plan) => RgbOutputScratch::YCbCr420, + ColorSpace::YCbCr if fancy_420 => RgbOutputScratch::YCbCr420, ColorSpace::YCbCr => RgbOutputScratch::YCbCrGeneric(ycbcr_generic_rows), ColorSpace::Rgb | ColorSpace::Cmyk | ColorSpace::Ycck => { RgbOutputScratch::RgbGeneric(rgb_generic_rows) diff --git a/crates/j2k-jpeg/src/entropy/sequential/generic/driver.rs b/crates/j2k-jpeg/src/entropy/sequential/generic/driver.rs index 9c7517957..00c635b0c 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/generic/driver.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/generic/driver.rs @@ -5,8 +5,8 @@ use super::super::emit::StripeEmit; use super::super::layout::{ decode_mcu_row_end_for_rect, expanded_output_rect, fast420_decode_mcu_row_end, - fast420_first_decode_mcu_row, first_decode_mcu_row_for_rect, is_ycbcr_420, - last_mcu_row_for_rect, mcu_row_intersects_rect, scaled_dimensions, stripe_region_layout, + fast420_first_decode_mcu_row, first_decode_mcu_row_for_rect, last_mcu_row_for_rect, + mcu_row_intersects_rect, scaled_dimensions, stripe_region_layout, uses_fancy_420_emit, StripeRegionLayout, }; use super::super::restart::{ @@ -15,6 +15,7 @@ use super::super::restart::{ use super::super::{PreparedDecodePlan, StripeBuffer}; use super::row::{decode_mcu_row, McuRowContext, McuRowState}; use crate::backend::Backend; +use crate::color::scaled_sampling::ScaledSampling; use crate::entropy::block::CoefficientBlock; use crate::error::{JpegError, Warning}; use crate::info::{DownscaleFactor, Rect}; @@ -30,6 +31,7 @@ pub(super) enum ScanOutputMode { #[derive(Clone, Copy)] pub(super) struct ScanSetup { + pub(super) scaled: ScaledSampling, block_size: u32, mcu_height_px: u32, mcus_per_row: u32, @@ -52,6 +54,7 @@ impl ScanSetup { mode: ScanOutputMode, ) -> Self { let (width, height) = scaled_dimensions(plan.dimensions, downscale); + let scaled = ScaledSampling::new(plan.sampling, downscale, plan.dimensions); let block_size = downscale.output_block_size(); let mcu_width_px = block_size * u32::from(plan.sampling.max_h); let mcu_height_px = block_size * u32::from(plan.sampling.max_v); @@ -61,7 +64,7 @@ impl ScanSetup { let full_output_rect = expanded_rect == Rect::full((width, height)); let use_420_context_window = matches!(mode, ScanOutputMode::InterleavedRgb) && !full_output_rect - && is_ycbcr_420(plan); + && uses_fancy_420_emit(plan, &scaled); let emit_rect = if use_420_context_window { output_rect } else { @@ -79,6 +82,7 @@ impl ScanSetup { }; Self { + scaled, block_size, mcu_height_px, mcus_per_row, @@ -101,6 +105,7 @@ impl ScanSetup { ) -> Result<(), JpegError> { pool.prepare_for( plan, + self.scaled.effective, self.region.stripe_mcus_per_row, self.block_size, plan.scratch_bytes, @@ -172,6 +177,7 @@ pub(super) fn decode_scan_rows( plan, backend, downscale, + scaled: &setup.scaled, output_rect: setup.expanded_rect, full_output_rect: setup.full_output_rect, stripe_mcu_start: setup.region.stripe_mcu_start, @@ -210,6 +216,7 @@ pub(super) fn decode_scan_rows( stripe_index: mcu_row - 1, source_width: setup.region.source_width_usize(), downscale, + scaled: &setup.scaled, })?; } core::mem::swap(&mut prev_stripe, &mut curr_stripe); @@ -227,6 +234,7 @@ pub(super) fn decode_scan_rows( stripe_index: current_mcu_row, source_width: setup.region.source_width_usize(), downscale, + scaled: &setup.scaled, })?; } finish_scan(&mut br, setup.decode_mcu_row_end == setup.mcu_rows) diff --git a/crates/j2k-jpeg/src/entropy/sequential/generic/row.rs b/crates/j2k-jpeg/src/entropy/sequential/generic/row.rs index 7a18ec38c..61928858d 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/generic/row.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/generic/row.rs @@ -10,6 +10,7 @@ use super::super::layout::{component_block_intersects_rect, ComponentBlockPositi use super::super::restart::{consume_restart_marker_if_due, McuPosition}; use super::super::{PreparedDecodePlan, StripeBuffer}; use crate::backend::Backend; +use crate::color::scaled_sampling::ScaledSampling; use crate::entropy::block::{ decode_block_with_activity, skip_block, BlockActivity, CoefficientBlock, }; @@ -22,6 +23,9 @@ pub(super) struct McuRowContext<'a> { pub(super) plan: &'a PreparedDecodePlan, pub(super) backend: Backend, pub(super) downscale: DownscaleFactor, + /// Per-component IDCT sizes; planes are laid out for its effective + /// sampling. + pub(super) scaled: &'a ScaledSampling, pub(super) output_rect: Rect, pub(super) full_output_rect: bool, pub(super) stripe_mcu_start: u32, @@ -89,7 +93,6 @@ fn decode_mcu_row_local( stripe: &mut StripeBuffer, ) -> Result<(), JpegError> { let stripe_mcu_end = context.stripe_mcu_start + context.stripe_mcus_per_row; - let block_size = context.downscale.output_block_size(); for comp in &context.plan.components { assert_stripe_deposit_capacity( stripe, @@ -97,7 +100,7 @@ fn decode_mcu_row_local( u32::from(comp.h), u32::from(comp.v), context.stripe_mcus_per_row, - block_size, + context.scaled.component(comp.output_index).idct_size, ); } let mut pixels_4x4 = [0u8; 16]; @@ -133,6 +136,9 @@ fn decode_mcu_row_local( let dc_table = tables.dc_table; let ac_table = tables.ac_table; let in_region = mx >= context.stripe_mcu_start && mx < stripe_mcu_end; + // libjpeg-turbo may decode a subsampled component with a larger + // reduced IDCT than the scale's own block size. + let block_size = context.scaled.component(plane_idx).idct_size; let local_mcu_x0_px = mx.saturating_sub(context.stripe_mcu_start) * u32::from(comp.h) * block_size; for vy in 0..u32::from(comp.v) { @@ -166,84 +172,51 @@ fn decode_mcu_row_local( )?; let block_x = local_mcu_x0_px + vx * block_size; let block_y = vy * block_size; - match context.downscale { - DownscaleFactor::Full => { - match activity { - BlockActivity::DcOnly => { - crate::idct::idct_islow_dc_only( - state.coeff.dc_coeff(), - state.pixels, - ); - } - BlockActivity::BottomHalfZero => { - context.backend.idct_bottom_half_zero( - state.coeff.coefficients(), - state.pixels, - ); - } - BlockActivity::General => { - context - .backend - .idct(state.coeff.coefficients(), state.pixels); - } - } - deposit_block( - &mut stripe.planes[plane_idx], - stripe.plane_strides[plane_idx], - block_x, - block_y, - state.pixels, - ); + let plane = &mut stripe.planes[plane_idx]; + let stride = stripe.plane_strides[plane_idx]; + match (block_size, activity) { + (8, BlockActivity::DcOnly) => { + crate::idct::idct_islow_dc_only(state.coeff.dc_coeff(), state.pixels); + deposit_block(plane, stride, block_x, block_y, state.pixels); + } + (8, BlockActivity::BottomHalfZero) => { + context + .backend + .idct_bottom_half_zero(state.coeff.coefficients(), state.pixels); + deposit_block(plane, stride, block_x, block_y, state.pixels); + } + (8, BlockActivity::General) => { + context + .backend + .idct(state.coeff.coefficients(), state.pixels); + deposit_block(plane, stride, block_x, block_y, state.pixels); } - DownscaleFactor::Half => { - if activity == BlockActivity::DcOnly { - downscale::idct_islow_4x4_dc_only( - state.coeff.dc_coeff(), - &mut pixels_4x4, - ); - } else { - downscale::idct_islow_4x4( - state.coeff.coefficients(), - &mut pixels_4x4, - ); - } - deposit_block_4x4( - &mut stripe.planes[plane_idx], - stripe.plane_strides[plane_idx], - block_x, - block_y, - &pixels_4x4, + (4, BlockActivity::DcOnly) => { + downscale::idct_islow_4x4_dc_only( + state.coeff.dc_coeff(), + &mut pixels_4x4, ); + deposit_block_4x4(plane, stride, block_x, block_y, &pixels_4x4); } - DownscaleFactor::Quarter => { - if activity == BlockActivity::DcOnly { - downscale::idct_islow_2x2_dc_only( - state.coeff.dc_coeff(), - &mut pixels_2x2, - ); - } else { - downscale::idct_islow_2x2( - state.coeff.coefficients(), - &mut pixels_2x2, - ); - } - deposit_block_2x2( - &mut stripe.planes[plane_idx], - stripe.plane_strides[plane_idx], - block_x, - block_y, - pixels_2x2, + (4, _) => { + downscale::idct_islow_4x4(state.coeff.coefficients(), &mut pixels_4x4); + deposit_block_4x4(plane, stride, block_x, block_y, &pixels_4x4); + } + (2, BlockActivity::DcOnly) => { + downscale::idct_islow_2x2_dc_only( + state.coeff.dc_coeff(), + &mut pixels_2x2, ); + deposit_block_2x2(plane, stride, block_x, block_y, pixels_2x2); + } + (2, _) => { + downscale::idct_islow_2x2(state.coeff.coefficients(), &mut pixels_2x2); + deposit_block_2x2(plane, stride, block_x, block_y, pixels_2x2); } - DownscaleFactor::Eighth => { + _ => { + debug_assert_eq!(block_size, 1, "IDCT sizes are 8, 4, 2 or 1"); let pixel = downscale::idct_islow_1x1(state.coeff.coefficients()); - deposit_block_1x1( - &mut stripe.planes[plane_idx], - stripe.plane_strides[plane_idx], - block_x, - block_y, - pixel, - ); + deposit_block_1x1(plane, stride, block_x, block_y, pixel); } } } diff --git a/crates/j2k-jpeg/src/entropy/sequential/layout.rs b/crates/j2k-jpeg/src/entropy/sequential/layout.rs index 3790a55fd..fced3c9e0 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/layout.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/layout.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: MIT OR Apache-2.0 use super::{PreparedComponentPlan, PreparedDecodePlan}; +use crate::color::scaled_sampling::ScaledSampling; use crate::info::{ColorSpace, DownscaleFactor, Rect}; #[derive(Clone, Copy)] @@ -209,6 +210,13 @@ impl Fast420RegionLayout { } } +/// Whether the dedicated 4:2:0 smoothing emitters produce this decode: coded +/// 4:2:0 YCbCr at a scale where libjpeg-turbo still upsamples its chroma with +/// `h2v2_fancy_upsample`. +pub(super) fn uses_fancy_420_emit(plan: &PreparedDecodePlan, scaled: &ScaledSampling) -> bool { + is_ycbcr_420(plan) && scaled.is_fancy_420() +} + pub(super) fn is_ycbcr_420(plan: &PreparedDecodePlan) -> bool { plan.color_space == ColorSpace::YCbCr && plan.sampling.max_h == 2 diff --git a/crates/j2k-jpeg/src/entropy/sequential/plan.rs b/crates/j2k-jpeg/src/entropy/sequential/plan.rs index fbd179d8a..11fab9056 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/plan.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/plan.rs @@ -64,8 +64,13 @@ impl PreparedDecodePlan { self.restart_interval.is_none() && self.matches_fast_420_device_shape() } + /// Coded 4:2:0 YCbCr whose full-size chroma libjpeg-turbo smooths with + /// `h2v2_fancy_upsample`, the only upsampler the fused 4:2:0 routes + /// implement. Chroma at most two samples wide (images at most four pixels + /// wide) is replicated instead and takes the generic route. pub(crate) fn matches_fast_420_device_shape(&self) -> bool { - is_ycbcr_420(self) + self.dimensions.0 > 4 + && is_ycbcr_420(self) && self.components.len() == 3 && self.components[0].output_index == 0 && self.components[0].h == 2 @@ -92,8 +97,12 @@ impl PreparedDecodePlan { && self.components[2].v == 1 } + /// Coded 4:2:2 YCbCr whose full-size chroma libjpeg-turbo smooths (more + /// than two chroma samples wide), the shape the device 4:2:2 kernels + /// implement. pub(crate) fn matches_fast_rgb422_shape(&self) -> bool { - self.color_space == ColorSpace::YCbCr + self.dimensions.0 > 4 + && self.color_space == ColorSpace::YCbCr && self.components.len() == 3 && self.components[0].output_index == 0 && self.components[0].h == 2 diff --git a/crates/j2k-jpeg/src/entropy/sequential/rgb444.rs b/crates/j2k-jpeg/src/entropy/sequential/rgb444.rs index e002f1c1c..c934e205d 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/rgb444.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/rgb444.rs @@ -48,6 +48,7 @@ pub(crate) fn decode_scan_fast_rgb_444( pool.prepare_for( plan, + plan.sampling, mcus_per_row, DownscaleFactor::Full.output_block_size(), plan.scratch_bytes, diff --git a/crates/j2k-jpeg/src/entropy/sequential/stripe.rs b/crates/j2k-jpeg/src/entropy/sequential/stripe.rs index da3514117..c82376c41 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/stripe.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/stripe.rs @@ -11,8 +11,6 @@ use crate::allocation::{ use crate::error::JpegError; use crate::info::SamplingFactors; -use super::PreparedDecodePlan; - #[derive(Debug, Default)] pub(crate) struct StripeBuffer { pub(crate) planes: Vec>, @@ -29,14 +27,6 @@ pub(crate) struct StripeLayout { } impl StripeLayout { - pub(crate) fn for_plan( - plan: &PreparedDecodePlan, - mcus_per_row: u32, - block_size: u32, - ) -> Result { - Self::for_sampling(plan.sampling, mcus_per_row, block_size) - } - pub(crate) fn for_sampling( sampling: SamplingFactors, mcus_per_row: u32, diff --git a/crates/j2k-jpeg/src/entropy/sequential/tests.rs b/crates/j2k-jpeg/src/entropy/sequential/tests.rs index 49c2a53fd..0137d483d 100644 --- a/crates/j2k-jpeg/src/entropy/sequential/tests.rs +++ b/crates/j2k-jpeg/src/entropy/sequential/tests.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: MIT OR Apache-2.0 use super::*; +use crate::color::scaled_sampling::ScaledSampling; use crate::entropy::huffman::{HuffmanTable, PreparedHuffmanTables}; use crate::info::{ColorSpace, SamplingFactors}; use crate::output::Rgb8Writer; @@ -422,13 +423,19 @@ fn emit_final_420_stripe( let block = downscale.output_block_size() as usize; let (scaled_width, scaled_height) = scaled_dimensions((13, height), downscale); let width = scaled_width as usize; - let real_chroma_rows = (scaled_height as usize - 2 * block).div_ceil(2); - let prev = synthetic_420_stripe(block, 5, None); - let curr = synthetic_420_stripe(block, 40, Some((real_chroma_rows, padding))); + let sampling = SamplingFactors::from_validated_components(&[(2, 2), (1, 1), (1, 1)]); + // Below full size the chroma is decoded at luma resolution (libjpeg-turbo + // enlarges its IDCT), so its planes match the luma plane. + let scaled = ScaledSampling::new(sampling, downscale, (13, height)); + let chroma = scaled.component(1); + let chroma_block = chroma.idct_size as usize; + let real_chroma_rows = (scaled_height as usize - 2 * block).div_ceil(chroma.v_ratio as usize); + let prev = synthetic_420_stripe(block, chroma_block, 5, None); + let curr = synthetic_420_stripe(block, chroma_block, 40, Some((real_chroma_rows, padding))); let plan = PreparedDecodePlan { components: vec![], huffman_tables: PreparedHuffmanTables::try_with_capacity(0).expect("empty test arena"), - sampling: SamplingFactors::from_validated_components(&[(2, 2), (1, 1), (1, 1)]), + sampling, color_space, restart_interval: None, dimensions: (13, height), @@ -447,6 +454,11 @@ fn emit_final_420_stripe( cr_top: vec![0; width], cr_bot: vec![0; width], }; + let mut ycbcr_rows = crate::internal::scratch::YCbCrGenericRows { + cb_up: vec![0; width], + cr_up: vec![0; width], + }; + let fancy_420 = scaled.is_fancy_420(); let mut out = vec![0u8; width * scaled_height as usize * 3]; let mut writer = Rgb8Writer::new(&mut out, width * 3, scaled_width); let emit = super::emit::StripeEmit { @@ -456,15 +468,23 @@ fn emit_final_420_stripe( stripe_index: 1, source_width: width, downscale, + scaled: &scaled, }; let result = match (rgb_emitter, color_space) { - (true, ColorSpace::YCbCr) => super::emit::emit_stripe_rgb( + (true, ColorSpace::YCbCr) if fancy_420 => super::emit::emit_stripe_rgb( &plan, Backend::detect(), &mut writer, &mut RgbOutputScratch::YCbCr420, emit, ), + (true, ColorSpace::YCbCr) => super::emit::emit_stripe_rgb( + &plan, + Backend::detect(), + &mut writer, + &mut RgbOutputScratch::YCbCrGeneric(&mut ycbcr_rows), + emit, + ), (true, _) => super::emit::emit_stripe_rgb( &plan, Backend::detect(), @@ -472,12 +492,18 @@ fn emit_final_420_stripe( &mut RgbOutputScratch::RgbGeneric(&mut rgb_rows), emit, ), - (false, ColorSpace::YCbCr) => super::emit::emit_stripe( + (false, ColorSpace::YCbCr) if fancy_420 => super::emit::emit_stripe( &plan, &mut writer, &mut OutputScratch::YCbCr420(&mut ycbcr420_rows), emit, ), + (false, ColorSpace::YCbCr) => super::emit::emit_stripe( + &plan, + &mut writer, + &mut OutputScratch::YCbCrGeneric(&mut ycbcr_rows), + emit, + ), (false, _) => super::emit::emit_stripe( &plan, &mut writer, @@ -489,16 +515,18 @@ fn emit_final_420_stripe( out.split_off(width * 2 * block * 3) } -/// One-MCU-wide 4:2:0 stripe of patterned samples. With -/// `Some((real_rows, padding))`, chroma rows from `real_rows` on hold -/// `padding`, or repeat the last real row when `padding` is `None`. +/// One-MCU-wide 4:2:0 stripe of patterned samples, with chroma decoded at +/// `chroma_block` samples per block. With `Some((real_rows, padding))`, chroma +/// rows from `real_rows` on hold `padding`, or repeat the last real row when +/// `padding` is `None`. fn synthetic_420_stripe( block: usize, + chroma_block: usize, seed: usize, chroma_padding: Option<(usize, Option)>, ) -> StripeBuffer { - let strides = [2 * block, block, block]; - let rows = [2 * block, block, block]; + let strides = [2 * block, chroma_block, chroma_block]; + let rows = [2 * block, chroma_block, chroma_block]; let planes = (0..3) .map(|index| { (0..strides[index] * rows[index]) diff --git a/crates/j2k-jpeg/src/idct/downscale.rs b/crates/j2k-jpeg/src/idct/downscale.rs index 149a7e284..8b8b707cf 100644 --- a/crates/j2k-jpeg/src/idct/downscale.rs +++ b/crates/j2k-jpeg/src/idct/downscale.rs @@ -206,8 +206,106 @@ fn idct_2x2_row(work: &[Wrapping; 16], output: &mut [u8; 4], row: usize) { output[out + 1] = descale_and_clamp(tmp10 - tmp0, shift); } +/// 12-bit `jidctred.c`: `BITS_IN_JSAMPLE == 12` keeps `PASS1_BITS = 1` for +/// headroom and computes in 64-bit `JLONG`. The C code's zero-AC shortcuts give +/// the same results as the full computation and are omitted, as in +/// [`super::scalar::idct_islow_12bit`]. +const PASS1_BITS_12: usize = 1; + +/// Reduced 4x4 IDCT of a 12-bit block (`jpeg_idct_4x4`). +pub(crate) fn idct_islow_12bit_4x4(input: &[i16; 64], output: &mut [u16; 16]) { + let fix = |value: Wrapping| i64::from(value.0); + let odd = |z1: i64, z2: i64, z3: i64, z4: i64| { + ( + z1 * -fix(FIX_0_211164243) + + z2 * fix(FIX_1_451774981) + + z3 * -fix(FIX_2_172734803) + + z4 * fix(FIX_1_061594337), + z1 * -fix(FIX_0_509795579) + + z2 * -fix(FIX_0_601344887) + + z3 * fix(FIX_0_899976223) + + z4 * fix(FIX_2_562915447), + ) + }; + let even = |p0: i64, p2: i64, p6: i64| { + let tmp0 = p0 << (CONST_BITS + 1); + let tmp2 = p2 * fix(FIX_1_847759065) - p6 * fix(FIX_0_765366865); + (tmp0 + tmp2, tmp0 - tmp2) + }; + let mut work = [0i64; 32]; + for col in (0..8).filter(|&col| col != 4) { + let p = |row: usize| i64::from(input[row * 8 + col]); + let (tmp10, tmp12) = even(p(0), p(2), p(6)); + let (tmp0, tmp2) = odd(p(7), p(5), p(3), p(1)); + let shift = CONST_BITS - PASS1_BITS_12 + 1; + work[col] = descale_i64(tmp10 + tmp2, shift); + work[24 + col] = descale_i64(tmp10 - tmp2, shift); + work[8 + col] = descale_i64(tmp12 + tmp0, shift); + work[16 + col] = descale_i64(tmp12 - tmp0, shift); + } + for row in 0..4 { + let w = |col: usize| work[row * 8 + col]; + let (tmp10, tmp12) = even(w(0), w(2), w(6)); + let (tmp0, tmp2) = odd(w(7), w(5), w(3), w(1)); + let shift = CONST_BITS + PASS1_BITS_12 + 3 + 1; + let out = &mut output[row * 4..row * 4 + 4]; + out[0] = level_shift_12bit(descale_i64(tmp10 + tmp2, shift)); + out[3] = level_shift_12bit(descale_i64(tmp10 - tmp2, shift)); + out[1] = level_shift_12bit(descale_i64(tmp12 + tmp0, shift)); + out[2] = level_shift_12bit(descale_i64(tmp12 - tmp0, shift)); + } +} + +/// Reduced 2x2 IDCT of a 12-bit block (`jpeg_idct_2x2`). +pub(crate) fn idct_islow_12bit_2x2(input: &[i16; 64], output: &mut [u16; 4]) { + let fix = |value: Wrapping| i64::from(value.0); + let odd = |p7: i64, p5: i64, p3: i64, p1: i64| { + p7 * -fix(FIX_0_720959822) + + p5 * fix(FIX_0_850430095) + + p3 * -fix(FIX_1_272758580) + + p1 * fix(FIX_3_624509785) + }; + let mut work = [0i64; 16]; + for col in [0, 1, 3, 5, 7] { + let p = |row: usize| i64::from(input[row * 8 + col]); + let tmp10 = p(0) << (CONST_BITS + 2); + let tmp0 = odd(p(7), p(5), p(3), p(1)); + let shift = CONST_BITS - PASS1_BITS_12 + 2; + work[col] = descale_i64(tmp10 + tmp0, shift); + work[8 + col] = descale_i64(tmp10 - tmp0, shift); + } + for row in 0..2 { + let w = |col: usize| work[row * 8 + col]; + let tmp10 = w(0) << (CONST_BITS + 2); + let tmp0 = odd(w(7), w(5), w(3), w(1)); + let shift = CONST_BITS + PASS1_BITS_12 + 3 + 2; + output[row * 2] = level_shift_12bit(descale_i64(tmp10 + tmp0, shift)); + output[row * 2 + 1] = level_shift_12bit(descale_i64(tmp10 - tmp0, shift)); + } +} + +/// Reduced 1x1 IDCT of a 12-bit block (`jpeg_idct_1x1`). +pub(crate) fn idct_islow_12bit_1x1(input: &[i16; 64]) -> u16 { + level_shift_12bit(descale_i64(i64::from(input[0]), 3)) +} + +const fn descale_i64(value: i64, shift: usize) -> i64 { + (value + (1 << (shift - 1))) >> shift +} + +#[expect( + clippy::cast_sign_loss, + reason = "reduced 12-bit IDCT samples are clamped to 0..=4095 before conversion" +)] +fn level_shift_12bit(value: i64) -> u16 { + (value + 2048).clamp(0, 4095) as u16 +} + +/// libjpeg's `DESCALE`: arithmetic right shift that rounds half up. Every +/// stage of `jidctred.c` (and libjpeg-turbo's NEON/SSE2 ports) rounds, so a +/// plain shift biases scaled output low by up to one level per component. fn descale(value: Wrapping, shift: usize) -> Wrapping { - Wrapping(value.0 >> shift) + Wrapping((value + Wrapping(1 << (shift - 1))).0 >> shift) } #[expect( @@ -215,7 +313,7 @@ fn descale(value: Wrapping, shift: usize) -> Wrapping { reason = "reduced IDCT samples are clamped to the u8 output range before conversion" )] fn descale_and_clamp(value: Wrapping, shift: usize) -> u8 { - let shifted = value.0 >> shift; + let shifted = descale(value, shift).0; let level_shifted = shifted.wrapping_add(128); level_shifted.clamp(0, 255) as u8 } @@ -255,4 +353,45 @@ mod tests { assert_eq!(actual2, expected2); } } + + #[test] + fn twelve_bit_reduced_idcts_of_dc_blocks_are_uniform_rounded_samples() { + // DESCALE(dc, 3) + 2048: the DC-only result every reduced size shares. + for (dc, expected) in [ + (4i16, 2049u16), + (-4, 2048), + (12, 2050), + (-5, 2047), + (32767, 4095), + ] { + let mut input = [0i16; 64]; + input[0] = dc; + let mut out4 = [0u16; 16]; + let mut out2 = [0u16; 4]; + idct_islow_12bit_4x4(&input, &mut out4); + idct_islow_12bit_2x2(&input, &mut out2); + assert_eq!(idct_islow_12bit_1x1(&input), expected, "1x1 dc={dc}"); + assert!(out4.iter().all(|&px| px == expected), "4x4 dc={dc}"); + assert!(out2.iter().all(|&px| px == expected), "2x2 dc={dc}"); + } + } + + #[test] + fn reduced_idcts_round_dc_half_up_like_libjpeg_descale() { + // jpeg_idct_1x1 and the DC-only paths of jpeg_idct_{4x4,2x2} compute + // DESCALE(dc, 3) = (dc + 4) >> 3. A plain shift gives 128, 127, 129 for + // the first three inputs. + for (dc, expected) in [(4i16, 129u8), (-4, 128), (12, 130), (-5, 127), (3, 128)] { + let mut input = [0i16; 64]; + input[0] = dc; + let mut out4 = [0u8; 16]; + let mut out2 = [0u8; 4]; + idct_islow_4x4(&input, &mut out4); + idct_islow_2x2(&input, &mut out2); + + assert_eq!(idct_islow_1x1(&input), expected, "1x1 dc={dc}"); + assert!(out4.iter().all(|&px| px == expected), "4x4 dc={dc}"); + assert!(out2.iter().all(|&px| px == expected), "2x2 dc={dc}"); + } + } } diff --git a/crates/j2k-jpeg/src/internal/scratch.rs b/crates/j2k-jpeg/src/internal/scratch.rs index 3d19a17ba..f3ca0fe47 100644 --- a/crates/j2k-jpeg/src/internal/scratch.rs +++ b/crates/j2k-jpeg/src/internal/scratch.rs @@ -19,6 +19,7 @@ use crate::allocation::{ }; use crate::entropy::sequential::{PreparedDecodePlan, StripeBuffer, StripeLayout}; use crate::error::JpegError; +use crate::info::SamplingFactors; use alloc::vec::Vec; use j2k_core::ScratchPool as CoreScratchPool; @@ -79,12 +80,13 @@ struct SequentialScratchLayout { impl SequentialScratchLayout { fn for_plan( plan: &PreparedDecodePlan, + sampling: SamplingFactors, mcus_per_row: u32, block_size: u32, ) -> Result { Ok(Self { - stripe: StripeLayout::for_plan(plan, mcus_per_row, block_size)?, - component_count: plan.sampling.len(), + stripe: StripeLayout::for_sampling(sampling, mcus_per_row, block_size)?, + component_count: sampling.len(), row_width: plan.dimensions.0.div_ceil(8 / block_size.max(1)) as usize, }) } @@ -117,15 +119,18 @@ impl ScratchPool { } /// Grow every internal scratch buffer to the shape required by `plan` - /// and zero the predictor so each decode starts clean. + /// and zero the predictor so each decode starts clean. `sampling` sizes + /// the component planes: the coded factors, or the effective factors of a + /// DCT-scaled decode (see `ScaledSampling`). pub(crate) fn prepare_for( &mut self, plan: &PreparedDecodePlan, + sampling: SamplingFactors, mcus_per_row: u32, block_size: u32, max_bytes: usize, ) -> Result<(), JpegError> { - let layout = SequentialScratchLayout::for_plan(plan, mcus_per_row, block_size)?; + let layout = SequentialScratchLayout::for_plan(plan, sampling, mcus_per_row, block_size)?; let target_bytes = layout.minimum_bytes(self.detached_sink_bytes)?; ensure_request_bytes(target_bytes, max_bytes)?; if self.retained_bytes() > target_bytes diff --git a/crates/j2k-jpeg/tests/decode_into.rs b/crates/j2k-jpeg/tests/decode_into.rs index 46651b8d9..0fdfdaba0 100644 --- a/crates/j2k-jpeg/tests/decode_into.rs +++ b/crates/j2k-jpeg/tests/decode_into.rs @@ -313,8 +313,7 @@ fn assert_rgb16_decode_case(case: &Rgb16DecodeCase) { let scaled_roi = scaled_rect_covering_for_test(case.roi, 2); let scaled_row_bytes = scaled_roi.w as usize * PixelFormat::Rgb16.bytes_per_pixel(); let stride = scaled_row_bytes + 6; - let expected_scaled = - expected_scaled_rgb16_pixels(&case.expected, case.dimensions.0 as usize, case.roi, 2); + let expected_scaled = full_scaled_rgb16_crop(&decoder, case.roi, Downscale::Half); let mut region_scaled = vec![0xaa; stride * scaled_roi.h as usize]; let outcome = decoder .decode_region_scaled_into( @@ -934,7 +933,12 @@ fn decode_region_scaled_into_rgba16_projects_progressive12_color_samples() { let scaled_roi = scaled_rect_covering_for_test(roi, 2); let row_bytes = scaled_roi.w as usize * PixelFormat::Rgba16.bytes_per_pixel(); let stride = row_bytes + 8; - let expected_rgb = expected_scaled_rgb16_pixels(&full, full_width, roi, 2); + assert_eq!( + full.len(), + full_width * dec.info().dimensions.1 as usize * 6, + "{label} reference geometry" + ); + let expected_rgb = full_scaled_rgb16_crop(&dec, roi, Downscale::Half); let expected = rgb16_to_rgba16(&expected_rgb, u16::MAX); let mut buf = vec![0xaau8; stride * scaled_roi.h as usize]; @@ -1566,6 +1570,28 @@ fn jpeg_rect(rect: PixelRect) -> Rect { } } +/// The region of a full-image scaled `Rgb16` decode that a region-scaled +/// decode of `roi` must reproduce. +/// +/// libjpeg-turbo parity of the scaled image itself is covered by +/// `tests/scaled_matrix.rs`; the hand-built 12-bit 4:2:2 and 4:2:0 fixtures +/// here use a Huffman table libjpeg-turbo rejects, so they check that region +/// routes agree with the full-image route. +fn full_scaled_rgb16_crop(decoder: &Decoder<'_>, roi: Rect, scale: Downscale) -> Vec { + let (full, _) = decoder + .decode_request(DecodeRequest::scaled(PixelFormat::Rgb16, scale)) + .expect("full-image scaled Rgb16 decode"); + let width = decoder.info().dimensions.0.div_ceil(scale.denominator()); + crop_rgb16_bytes( + &full, + width as usize, + scaled_rect_covering_for_test(roi, scale.denominator()), + ) +} + +/// Point samples of a full-size decode at the scaled grid. Equal to a DCT +/// scaled decode only for fixtures whose blocks are uniform (DC-only with +/// equal neighbours), which is what the remaining callers use. fn expected_scaled_rgb16_pixels(full: &[u8], full_width: usize, roi: Rect, denom: u32) -> Vec { let scaled = scaled_rect_covering_for_test(roi, denom); let mut expected = Vec::with_capacity(scaled.w as usize * scaled.h as usize * 6); diff --git a/crates/j2k-jpeg/tests/scaled_matrix.rs b/crates/j2k-jpeg/tests/scaled_matrix.rs new file mode 100644 index 000000000..41e762743 --- /dev/null +++ b/crates/j2k-jpeg/tests/scaled_matrix.rs @@ -0,0 +1,565 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 + +//! DCT-scaled decodes against the libjpeg-turbo reference matrix. +//! +//! libjpeg-turbo gives each component its own reduced IDCT size (chroma is +//! decoded larger so it needs less upsampling), replicates instead of +//! smoothing at 1/8 and for 2:1 components at most two samples wide, and +//! applies the reduced IDCT to progressive and 12-bit coefficients. `OpenSlide` +//! scale-decodes NDPI/VMS levels with it, so j2k must match it bit for bit. + +use std::panic::{catch_unwind, AssertUnwindSafe}; + +use j2k_jpeg::{ + decode_tiles_region_scaled_into, decode_tiles_scaled_into, DecodeOutcome, DecodeRequest, + Decoder, Downscale, JpegError, PixelFormat, Rect, TileBatchOptions, TileRegionScaledDecodeJob, + TileScaledDecodeJob, +}; +use j2k_test_support::{ + crop_interleaved_bytes, scaled_matrix_cases, scaled_rect_covering, PixelRect, ScaledMatrixCase, + SCALED_MATRIX_DENOMINATORS, +}; + +fn downscale(denominator: u32) -> Downscale { + match denominator { + 1 => Downscale::None, + 2 => Downscale::Half, + 4 => Downscale::Quarter, + 8 => Downscale::Eighth, + _ => unreachable!("matrix denominators are 1, 2, 4, 8"), + } +} + +fn pixel_format(case: &ScaledMatrixCase) -> PixelFormat { + match (case.is_gray(), case.precision > 8) { + (true, false) => PixelFormat::Gray8, + (false, false) => PixelFormat::Rgb8, + (true, true) => PixelFormat::Gray16, + (false, true) => PixelFormat::Rgb16, + } +} + +/// Reference samples as native values (8-bit bytes or 12-bit values). +fn reference_samples(case: &ScaledMatrixCase, denominator: u32) -> Vec { + if case.precision > 8 { + case.reference_u16(denominator) + } else { + case.reference(denominator) + .iter() + .map(|&v| u16::from(v)) + .collect() + } +} + +/// Decoder output bytes as native values (16-bit output is little-endian). +fn output_samples(case: &ScaledMatrixCase, bytes: &[u8]) -> Vec { + if case.precision > 8 { + bytes + .chunks_exact(2) + .map(|pair| u16::from_le_bytes([pair[0], pair[1]])) + .collect() + } else { + bytes.iter().map(|&v| u16::from(v)).collect() + } +} + +/// 12-bit layouts the decoder has never implemented at any scale. Their +/// references stay in the matrix so enabling them is checked the same way. +fn is_unimplemented_12bit_layout(case: &ScaledMatrixCase) -> bool { + case.precision > 8 && matches!(case.layout, "440" | "411" | "410" | "1x4") +} + +/// Decodes `request`, turning a panic into a reported failure so one bad case +/// cannot hide the rest of the matrix. +fn decode(decoder: &Decoder<'_>, request: DecodeRequest) -> Result, String> { + let outcome: Result<(Vec, DecodeOutcome), JpegError> = + match catch_unwind(AssertUnwindSafe(|| decoder.decode_request(request))) { + Ok(outcome) => outcome, + Err(panic) => { + let message = panic + .downcast_ref::() + .map(String::as_str) + .or_else(|| panic.downcast_ref::<&str>().copied()) + .unwrap_or("non-string panic"); + return Err(format!("decode panicked: {message}")); + } + }; + outcome + .map(|(bytes, _)| bytes) + .map_err(|err| format!("decode failed: {err}")) +} + +fn expect_not_implemented( + decoder: &Decoder<'_>, + request: DecodeRequest, + label: &str, + failures: &mut Vec, +) { + match catch_unwind(AssertUnwindSafe(|| decoder.decode_request(request))) { + Ok(Err(JpegError::NotImplemented { .. })) => {} + Ok(Err(err)) => failures.push(format!("{label}: expected NotImplemented, got {err}")), + Ok(Ok(_)) => failures.push(format!( + "{label}: now decodes; remove it from is_unimplemented_12bit_layout" + )), + Err(_) => failures.push(format!("{label}: decode panicked")), + } +} + +fn describe_mismatch(expected: &[u16], actual: &[u16]) -> Option { + if expected.len() != actual.len() { + return Some(format!( + "{} samples, expected {}", + actual.len(), + expected.len() + )); + } + let diffs: Vec = expected + .iter() + .zip(actual) + .map(|(&e, &a)| e.abs_diff(a)) + .filter(|&d| d > 0) + .collect(); + let max = diffs.iter().copied().max()?; + Some(format!( + "{} of {} samples differ, max {max}", + diffs.len(), + expected.len() + )) +} + +fn crop_samples(samples: &[u16], width: u32, channels: usize, rect: PixelRect) -> Vec { + let mut out = Vec::new(); + for y in rect.y..rect.y + rect.h { + let start = (y * width + rect.x) as usize * channels; + out.extend_from_slice(&samples[start..start + rect.w as usize * channels]); + } + out +} + +fn assert_no_failures(failures: &[String]) { + assert!( + failures.is_empty(), + "{} scaled decodes differ from libjpeg-turbo:\n{}", + failures.len(), + failures.join("\n") + ); +} + +#[test] +fn scaled_decodes_match_libjpeg_turbo() { + let mut failures = Vec::new(); + for case in scaled_matrix_cases() { + let decoder = match Decoder::new(case.jpeg) { + Ok(decoder) => decoder, + Err(err) => { + failures.push(format!("{}: parse failed: {err}", case.name)); + continue; + } + }; + for denominator in SCALED_MATRIX_DENOMINATORS { + let request = DecodeRequest::scaled(pixel_format(&case), downscale(denominator)); + let label = format!("{} 1/{denominator}", case.name); + if is_unimplemented_12bit_layout(&case) { + expect_not_implemented(&decoder, request, &label, &mut failures); + continue; + } + match decode(&decoder, request) { + Ok(bytes) => { + let expected = reference_samples(&case, denominator); + if let Some(message) = + describe_mismatch(&expected, &output_samples(&case, &bytes)) + { + failures.push(format!("{label}: {message}")); + } + } + Err(err) => failures.push(format!("{label}: {err}")), + } + } + } + assert_no_failures(&failures); +} + +#[test] +fn scaled_rgba_decodes_match_libjpeg_turbo() { + let mut failures = Vec::new(); + for case in scaled_matrix_cases() + .into_iter() + .filter(|case| !case.is_gray() && case.precision == 8) + { + let decoder = Decoder::new(case.jpeg).expect("matrix JPEG parses"); + for denominator in SCALED_MATRIX_DENOMINATORS { + let request = DecodeRequest::scaled(PixelFormat::Rgba8, downscale(denominator)); + let label = format!("{} rgba 1/{denominator}", case.name); + match decode(&decoder, request) { + Ok(bytes) => { + let rgb: Vec = bytes + .chunks_exact(4) + .flat_map(|px| px[..3].iter().map(|&v| u16::from(v))) + .collect(); + let alpha_ok = bytes.chunks_exact(4).all(|px| px[3] == u8::MAX); + if let Some(message) = + describe_mismatch(&reference_samples(&case, denominator), &rgb) + { + failures.push(format!("{label}: {message}")); + } else if !alpha_ok { + failures.push(format!("{label}: alpha is not opaque")); + } + } + Err(err) => failures.push(format!("{label}: {err}")), + } + } + } + assert_no_failures(&failures); +} + +/// Scaled regions are crops of the scaled image, wherever the region starts. +#[test] +fn scaled_region_decodes_match_libjpeg_turbo_crops() { + let mut failures = Vec::new(); + for case in scaled_matrix_cases() + .into_iter() + .filter(|case| case.width >= 18 || case.height >= 64) + { + let decoder = Decoder::new(case.jpeg).expect("matrix JPEG parses"); + let rois = [ + Rect { + x: 5.min(case.width / 3), + y: 3, + w: case.width - 9.min(case.width / 2), + h: case.height - 6, + }, + Rect { + x: case.width / 2 + 1, + y: case.height / 2 - 1, + w: case.width / 2 - 1, + h: case.height / 2, + }, + ]; + for roi in rois { + for denominator in SCALED_MATRIX_DENOMINATORS { + let request = + DecodeRequest::region_scaled(pixel_format(&case), roi, downscale(denominator)); + let label = format!( + "{} roi {},{} {}x{} 1/{denominator}", + case.name, roi.x, roi.y, roi.w, roi.h + ); + if is_unimplemented_12bit_layout(&case) { + expect_not_implemented(&decoder, request, &label, &mut failures); + continue; + } + let scaled = scaled_rect_covering( + PixelRect { + x: roi.x, + y: roi.y, + w: roi.w, + h: roi.h, + }, + denominator, + ); + let (width, _) = case.scaled_dimensions(denominator); + let expected = crop_samples( + &reference_samples(&case, denominator), + width, + case.channels(), + scaled, + ); + match decode(&decoder, request) { + Ok(bytes) => { + if let Some(message) = + describe_mismatch(&expected, &output_samples(&case, &bytes)) + { + failures.push(format!("{label}: {message}")); + } + } + Err(err) => failures.push(format!("{label}: {err}")), + } + } + } + } + assert_no_failures(&failures); +} + +#[test] +fn crop_helper_matches_interleaved_byte_crop() { + let samples: Vec = (0..60).collect(); + let rect = PixelRect { + x: 1, + y: 1, + w: 2, + h: 2, + }; + let widened: Vec = samples.iter().map(|&v| u16::from(v)).collect(); + let expected: Vec = crop_interleaved_bytes(&samples, 5, 3, rect) + .into_iter() + .map(u16::from) + .collect(); + assert_eq!(crop_samples(&widened, 5, 3, rect), expected); +} + +/// The batch tile API (what `wsi-rs` calls for NDPI/VMS scaled levels) routes +/// through the same scaled geometry as single decodes. +#[test] +fn batch_scaled_tile_decodes_match_libjpeg_turbo() { + let cases: Vec = scaled_matrix_cases() + .into_iter() + .filter(|case| case.precision == 8) + .collect(); + let mut failures = Vec::new(); + for denominator in SCALED_MATRIX_DENOMINATORS { + let mut outputs: Vec> = cases + .iter() + .map(|case| vec![0u8; case.reference(denominator).len() * 3 / case.channels()]) + .collect(); + let mut jobs: Vec> = cases + .iter() + .zip(outputs.iter_mut()) + .map(|(case, out)| TileScaledDecodeJob { + input: case.jpeg, + stride: case.scaled_dimensions(denominator).0 as usize * 3, + out: out.as_mut_slice(), + scale: downscale(denominator), + }) + .collect(); + // Grayscale cases decode to RGB here: replicate their references. + if let Err(err) = + decode_tiles_scaled_into(&mut jobs, PixelFormat::Rgb8, TileBatchOptions::default()) + { + failures.push(format!("1/{denominator} batch failed: {err}")); + } + drop(jobs); + for (case, out) in cases.iter().zip(&outputs) { + let expected = rgb_reference(case, denominator); + let actual: Vec = out.iter().map(|&v| u16::from(v)).collect(); + if let Some(message) = describe_mismatch(&expected, &actual) { + failures.push(format!("{} batch 1/{denominator}: {message}", case.name)); + } + } + } + assert_no_failures(&failures); +} + +#[test] +fn batch_scaled_region_tile_decodes_match_libjpeg_turbo_crops() { + let cases: Vec = scaled_matrix_cases() + .into_iter() + .filter(|case| case.precision == 8 && case.width >= 18) + .collect(); + let mut failures = Vec::new(); + for denominator in SCALED_MATRIX_DENOMINATORS { + let rois: Vec = cases + .iter() + .map(|case| Rect { + x: case.width / 3, + y: case.height / 4, + w: case.width / 2, + h: case.height / 2, + }) + .collect(); + let scaled: Vec = rois + .iter() + .map(|roi| { + scaled_rect_covering( + PixelRect { + x: roi.x, + y: roi.y, + w: roi.w, + h: roi.h, + }, + denominator, + ) + }) + .collect(); + let mut outputs: Vec> = scaled + .iter() + .map(|rect| vec![0u8; rect.w as usize * rect.h as usize * 3]) + .collect(); + let mut jobs: Vec> = cases + .iter() + .zip(&rois) + .zip(&scaled) + .zip(outputs.iter_mut()) + .map(|(((case, &roi), rect), out)| TileRegionScaledDecodeJob { + input: case.jpeg, + out: out.as_mut_slice(), + stride: rect.w as usize * 3, + roi: roi.into(), + scale: downscale(denominator), + }) + .collect(); + if let Err(err) = decode_tiles_region_scaled_into( + &mut jobs, + PixelFormat::Rgb8, + TileBatchOptions::default(), + ) { + failures.push(format!("1/{denominator} batch failed: {err}")); + } + drop(jobs); + for ((case, rect), out) in cases.iter().zip(&scaled).zip(&outputs) { + let (width, _) = case.scaled_dimensions(denominator); + let expected = crop_samples(&rgb_reference(case, denominator), width, 3, *rect); + let actual: Vec = out.iter().map(|&v| u16::from(v)).collect(); + if let Some(message) = describe_mismatch(&expected, &actual) { + failures.push(format!( + "{} batch region 1/{denominator}: {message}", + case.name + )); + } + } + } + assert_no_failures(&failures); +} + +/// 8-bit reference as interleaved RGB (grayscale expanded to R = G = B). +fn rgb_reference(case: &ScaledMatrixCase, denominator: u32) -> Vec { + let samples = reference_samples(case, denominator); + if case.is_gray() { + samples.iter().flat_map(|&v| [v, v, v]).collect() + } else { + samples + } +} + +/// Region edges on and beside MCU boundaries at every scale, where smoothing +/// filters need chroma from outside the decoded window. +fn edge_sweep_rois(width: u32, height: u32) -> Vec { + let xs = [0, 7, 8, 9, 15, 16, 17]; + let ys = [0, 7, 8, 9, 16, 17]; + let mut rois = Vec::new(); + for &x0 in &xs { + for &x1 in &xs { + for &y0 in &ys { + for &y1 in &ys { + let (x1, y1) = (x1 + 8, y1 + 8); + if x0 < x1 && y0 < y1 && x1 <= width && y1 <= height { + rois.push(Rect { + x: x0, + y: y0, + w: x1 - x0, + h: y1 - y0, + }); + } + } + } + } + } + rois +} + +#[test] +fn scaled_region_edge_sweep_matches_libjpeg_turbo_crops() { + let mut failures = Vec::new(); + for case in scaled_matrix_cases() + .into_iter() + .filter(|case| case.width == 45 && !is_unimplemented_12bit_layout(case)) + { + let decoder = Decoder::new(case.jpeg).expect("matrix JPEG parses"); + for denominator in SCALED_MATRIX_DENOMINATORS { + let reference = reference_samples(&case, denominator); + let (width, _) = case.scaled_dimensions(denominator); + for roi in edge_sweep_rois(case.width, case.height) { + let scaled = scaled_rect_covering( + PixelRect { + x: roi.x, + y: roi.y, + w: roi.w, + h: roi.h, + }, + denominator, + ); + let expected = crop_samples(&reference, width, case.channels(), scaled); + let request = + DecodeRequest::region_scaled(pixel_format(&case), roi, downscale(denominator)); + let label = format!( + "{} roi {},{} {}x{} 1/{denominator}", + case.name, roi.x, roi.y, roi.w, roi.h + ); + match decode(&decoder, request) { + Ok(bytes) => { + if let Some(message) = + describe_mismatch(&expected, &output_samples(&case, &bytes)) + { + failures.push(format!("{label}: {message}")); + } + } + Err(err) => failures.push(format!("{label}: {err}")), + } + } + } + } + assert_no_failures(&failures); +} + +/// Grayscale images decoded to RGB or RGBA replicate the gray sample into +/// every channel at every scale, for 8-bit and 12-bit precision. +#[test] +fn scaled_gray_to_color_decodes_match_libjpeg_turbo() { + let mut failures = Vec::new(); + for case in scaled_matrix_cases() + .into_iter() + .filter(ScaledMatrixCase::is_gray) + { + let decoder = Decoder::new(case.jpeg).expect("matrix JPEG parses"); + let formats = if case.precision > 8 { + [(PixelFormat::Rgb16, 3), (PixelFormat::Rgba16, 4)] + } else { + [(PixelFormat::Rgb8, 3), (PixelFormat::Rgba8, 4)] + }; + let full = Rect { + x: 0, + y: 0, + w: case.width, + h: case.height, + }; + let inner = Rect { + x: case.width / 3, + y: case.height / 3, + w: case.width - case.width / 3, + h: case.height - case.height / 3, + }; + for denominator in SCALED_MATRIX_DENOMINATORS { + let gray = reference_samples(&case, denominator); + let (width, _) = case.scaled_dimensions(denominator); + for roi in [full, inner] { + let scaled = scaled_rect_covering( + PixelRect { + x: roi.x, + y: roi.y, + w: roi.w, + h: roi.h, + }, + denominator, + ); + let crop = crop_samples(&gray, width, 1, scaled); + for (fmt, channels) in formats { + let alpha = if case.precision > 8 { u16::MAX } else { 255 }; + let expected: Vec = crop + .iter() + .flat_map(|&v| { + let mut px = vec![v; 3]; + if channels == 4 { + px.push(alpha); + } + px + }) + .collect(); + let request = DecodeRequest::region_scaled(fmt, roi, downscale(denominator)); + let label = format!( + "{} {fmt:?} roi {},{} {}x{} 1/{denominator}", + case.name, roi.x, roi.y, roi.w, roi.h + ); + match decode(&decoder, request) { + Ok(bytes) => { + if let Some(message) = + describe_mismatch(&expected, &output_samples(&case, &bytes)) + { + failures.push(format!("{label}: {message}")); + } + } + Err(err) => failures.push(format!("{label}: {err}")), + } + } + } + } + } + assert_no_failures(&failures); +} diff --git a/crates/j2k-jpeg/tests/wsi_parity.rs b/crates/j2k-jpeg/tests/wsi_parity.rs index 52e93a7d4..689c25cf7 100644 --- a/crates/j2k-jpeg/tests/wsi_parity.rs +++ b/crates/j2k-jpeg/tests/wsi_parity.rs @@ -7,7 +7,9 @@ use j2k_test_support::{ crop_interleaved_bytes, restart_coded_grayscale_jpeg, scaled_rect_covering, PixelRect, JPEG_BASELINE_420_16X16, JPEG_BASELINE_420_16X16_RGB, JPEG_BASELINE_420_RESTART_32X16, JPEG_BASELINE_420_RESTART_32X16_RGB, JPEG_BASELINE_422_16X8, JPEG_BASELINE_422_16X8_RGB, - JPEG_BASELINE_444_8X8, JPEG_BASELINE_444_8X8_RGB, JPEG_GRAYSCALE_8X8, JPEG_GRAYSCALE_8X8_GRAY, + JPEG_BASELINE_444_8X8, JPEG_BASELINE_444_8X8_RGB, JPEG_BASELINE_444_RESTART_67X45, + JPEG_BASELINE_444_RESTART_67X45_EIGHTH_RGB, JPEG_BASELINE_444_RESTART_67X45_HALF_RGB, + JPEG_BASELINE_444_RESTART_67X45_QUARTER_RGB, JPEG_GRAYSCALE_8X8, JPEG_GRAYSCALE_8X8_GRAY, }; const BASELINE_420_JPG: &[u8] = JPEG_BASELINE_420_16X16; @@ -16,6 +18,21 @@ const BASELINE_420_RGB: &[u8] = JPEG_BASELINE_420_16X16_RGB; const GRAYSCALE_8X8_JPG: &[u8] = JPEG_GRAYSCALE_8X8; const GRAYSCALE_8X8_GRAY: &[u8] = JPEG_GRAYSCALE_8X8_GRAY; +/// `djpeg -scale 1/N` references for the NDPI/VMS-shaped 4:4:4 restart fixture. +const BASELINE_444_RESTART_SCALED: [(Downscale, u32, &[u8]); 3] = [ + (Downscale::Half, 2, JPEG_BASELINE_444_RESTART_67X45_HALF_RGB), + ( + Downscale::Quarter, + 4, + JPEG_BASELINE_444_RESTART_67X45_QUARTER_RGB, + ), + ( + Downscale::Eighth, + 8, + JPEG_BASELINE_444_RESTART_67X45_EIGHTH_RGB, + ), +]; + #[test] fn baseline_420_16x16_matches_libjpeg_turbo_bit_exact() { let dec = Decoder::new(BASELINE_420_JPG).expect("fixture must parse"); @@ -122,6 +139,52 @@ fn baseline_422_roi_and_scaled_roi_match_full_route_projections() { } } +#[test] +fn baseline_444_restart_scaled_decode_matches_libjpeg_turbo_reduced_idct() { + let decoder = + Decoder::new(JPEG_BASELINE_444_RESTART_67X45).expect("4:4:4 restart fixture must parse"); + assert_eq!(decoder.info().dimensions, (67, 45)); + + for (factor, denominator, expected) in BASELINE_444_RESTART_SCALED { + let (actual, _) = decoder + .decode_request(DecodeRequest::scaled(PixelFormat::Rgb8, factor)) + .expect("scaled 4:4:4 decode must succeed"); + assert_eq!( + actual, expected, + "1/{denominator} output must match djpeg -scale 1/{denominator}" + ); + } +} + +#[test] +fn baseline_444_restart_scaled_region_matches_libjpeg_turbo_crop() { + let decoder = + Decoder::new(JPEG_BASELINE_444_RESTART_67X45).expect("4:4:4 restart fixture must parse"); + let roi = Rect { + x: 13, + y: 9, + w: 41, + h: 27, + }; + + for (factor, denominator, expected) in BASELINE_444_RESTART_SCALED { + let (region, outcome) = decoder + .decode_request(DecodeRequest::region_scaled(PixelFormat::Rgb8, roi, factor)) + .expect("scaled 4:4:4 ROI decode must succeed"); + let scaled_width = 67u32.div_ceil(denominator) as usize; + assert_eq!( + region, + crop_rgb8( + expected, + scaled_width, + scaled_rect_covering_by(roi, denominator) + ), + "1/{denominator} ROI must match the djpeg -scale 1/{denominator} crop" + ); + assert_eq!(outcome.decoded, roi); + } +} + #[test] fn grayscale_8x8_matches_libjpeg_turbo_bit_exact() { let dec = Decoder::new(GRAYSCALE_8X8_JPG).expect("grayscale fixture must parse"); @@ -174,7 +237,7 @@ fn baseline_420_wsi_shaped_scaled_region_matches_full_decode_crop() { .expect("scaled region decode must succeed") .0; - let scaled_roi = scaled_rect_covering_half(roi); + let scaled_roi = scaled_rect_covering_by(roi, 2); assert_eq!(region, crop_rgb8(&full, 8, scaled_roi)); } @@ -235,8 +298,8 @@ fn crop_gray8(full: &[u8], width: usize, roi: Rect) -> Vec { crop_interleaved_bytes(full, width, 1, pixel_rect(roi)) } -fn scaled_rect_covering_half(roi: Rect) -> Rect { - let scaled = scaled_rect_covering(pixel_rect(roi), 2); +fn scaled_rect_covering_by(roi: Rect, denominator: u32) -> Rect { + let scaled = scaled_rect_covering(pixel_rect(roi), denominator); Rect { x: scaled.x, y: scaled.y, diff --git a/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45.jpg b/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45.jpg new file mode 100644 index 000000000..88a73109e Binary files /dev/null and b/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45.jpg differ diff --git a/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45_eighth.rgb b/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45_eighth.rgb new file mode 100644 index 000000000..8b7ad6f25 Binary files /dev/null and b/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45_eighth.rgb differ diff --git a/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45_half.rgb b/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45_half.rgb new file mode 100644 index 000000000..45c607d51 Binary files /dev/null and b/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45_half.rgb differ diff --git a/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45_quarter.rgb b/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45_quarter.rgb new file mode 100644 index 000000000..4d552de32 Binary files /dev/null and b/crates/j2k-test-support/fixtures/conformance/baseline_444_restart_67x45_quarter.rgb differ diff --git a/crates/j2k-test-support/fixtures/scaled_matrix/cases.tsv b/crates/j2k-test-support/fixtures/scaled_matrix/cases.tsv new file mode 100644 index 000000000..f8f004427 --- /dev/null +++ b/crates/j2k-test-support/fixtures/scaled_matrix/cases.tsv @@ -0,0 +1,210 @@ +# Generated by corpus/conformance/scaled_matrix.py with libjpeg-turbo version 3.1.4.1 (build 20260327) +# name layout coding precision width height jpeg scale1 scale2 scale4 scale8 +gray_3x3_baseline_8bit gray baseline 8 3 3 15:196 0:9 9:4 13:1 14:1 +gray_3x3_restart_8bit gray restart 8 3 3 226:202 211:9 220:4 224:1 225:1 +gray_3x3_progressive_8bit gray progressive 8 3 3 443:319 428:9 437:4 441:1 442:1 +gray_4x4_baseline_8bit gray baseline 8 4 4 784:198 762:16 778:4 782:1 783:1 +gray_4x4_restart_8bit gray restart 8 4 4 1004:204 982:16 998:4 1002:1 1003:1 +gray_4x4_progressive_8bit gray progressive 8 4 4 1230:321 1208:16 1224:4 1228:1 1229:1 +gray_9x9_baseline_8bit gray baseline 8 9 9 1670:239 1551:81 1632:25 1657:9 1666:4 +gray_9x9_restart_8bit gray restart 8 9 9 2028:251 1909:81 1990:25 2015:9 2024:4 +gray_9x9_progressive_8bit gray progressive 8 9 9 2398:367 2279:81 2360:25 2385:9 2394:4 +gray_18x14_baseline_8bit gray baseline 8 18 14 3106:348 2765:252 3017:63 3080:20 3100:6 +gray_18x14_restart_8bit gray restart 8 18 14 3795:365 3454:252 3706:63 3769:20 3789:6 +gray_18x14_progressive_8bit gray progressive 8 18 14 4501:471 4160:252 4412:63 4475:20 4495:6 +gray_45x31_baseline_8bit gray baseline 8 45 31 6855:911 4972:1395 6367:368 6735:96 6831:24 +gray_45x31_restart_8bit gray restart 8 45 31 9649:977 7766:1395 9161:368 9529:96 9625:24 +gray_45x31_progressive_8bit gray progressive 8 45 31 12509:1014 10626:1395 12021:368 12389:96 12485:24 +gray_67x45_baseline_8bit gray baseline 8 67 45 17578:1784 13523:3015 16538:782 17320:204 17524:54 +gray_67x45_restart_8bit gray restart 8 67 45 23417:1919 19362:3015 22377:782 23159:204 23363:54 +gray_67x45_progressive_8bit gray progressive 8 67 45 29391:1839 25336:3015 28351:782 29133:204 29337:54 +444_3x3_baseline_8bit 444 baseline 8 3 3 31275:380 31230:27 31257:12 31269:3 31272:3 +444_3x3_restart_8bit 444 restart 8 3 3 31700:386 31655:27 31682:12 31694:3 31697:3 +444_3x3_progressive_8bit 444 progressive 8 3 3 32131:625 32086:27 32113:12 32125:3 32128:3 +444_4x4_baseline_8bit 444 baseline 8 4 4 32822:379 32756:48 32804:12 32816:3 32819:3 +444_4x4_restart_8bit 444 restart 8 4 4 33267:385 33201:48 33249:12 33261:3 33264:3 +444_4x4_progressive_8bit 444 progressive 8 4 4 33718:621 33652:48 33700:12 33712:3 33715:3 +444_9x9_baseline_8bit 444 baseline 8 9 9 34696:490 34339:243 34582:75 34657:27 34684:12 +444_9x9_restart_8bit 444 restart 8 9 9 35543:503 35186:243 35429:75 35504:27 35531:12 +444_9x9_progressive_8bit 444 progressive 8 9 9 36403:749 36046:243 36289:75 36364:27 36391:12 +444_18x14_baseline_8bit 444 baseline 8 18 14 38175:764 37152:756 37908:189 38097:60 38157:18 +444_18x14_restart_8bit 444 restart 8 18 14 39962:781 38939:756 39695:189 39884:60 39944:18 +444_18x14_progressive_8bit 444 progressive 8 18 14 41766:1011 40743:756 41499:189 41688:60 41748:18 +444_45x31_baseline_8bit 444 baseline 8 45 31 48426:2207 42777:4185 46962:1104 48066:288 48354:72 +444_45x31_restart_8bit 444 restart 8 45 31 56282:2273 50633:4185 54818:1104 55922:288 56210:72 +444_45x31_progressive_8bit 444 progressive 8 45 31 64204:2439 58555:4185 62740:1104 63844:288 64132:72 +444_67x45_baseline_8bit 444 baseline 8 67 45 78808:4436 66643:9045 75688:2346 78034:612 78646:162 +444_67x45_restart_8bit 444 restart 8 67 45 95409:4577 83244:9045 92289:2346 94635:612 95247:162 +444_67x45_progressive_8bit 444 progressive 8 67 45 112151:4603 99986:9045 109031:2346 111377:612 111989:162 +422_3x3_baseline_8bit 422 baseline 8 3 3 116799:363 116754:27 116781:12 116793:3 116796:3 +422_3x3_restart_8bit 422 restart 8 3 3 117207:369 117162:27 117189:12 117201:3 117204:3 +422_3x3_progressive_8bit 422 progressive 8 3 3 117621:600 117576:27 117603:12 117615:3 117618:3 +422_4x4_baseline_8bit 422 baseline 8 4 4 118287:369 118221:48 118269:12 118281:3 118284:3 +422_4x4_restart_8bit 422 restart 8 4 4 118722:375 118656:48 118704:12 118716:3 118719:3 +422_4x4_progressive_8bit 422 progressive 8 4 4 119163:610 119097:48 119145:12 119157:3 119160:3 +422_9x9_baseline_8bit 422 baseline 8 9 9 120130:457 119773:243 120016:75 120091:27 120118:12 +422_9x9_restart_8bit 422 restart 8 9 9 120944:469 120587:243 120830:75 120905:27 120932:12 +422_9x9_progressive_8bit 422 progressive 8 9 9 121770:710 121413:243 121656:75 121731:27 121758:12 +422_18x14_baseline_8bit 422 baseline 8 18 14 123503:636 122480:756 123236:189 123425:60 123485:18 +422_18x14_restart_8bit 422 restart 8 18 14 125162:648 124139:756 124895:189 125084:60 125144:18 +422_18x14_progressive_8bit 422 progressive 8 18 14 126833:889 125810:756 126566:189 126755:60 126815:18 +422_45x31_baseline_8bit 422 baseline 8 45 31 133371:1572 127722:4185 131907:1104 133011:288 133299:72 +422_45x31_restart_8bit 422 restart 8 45 31 140592:1610 134943:4185 139128:1104 140232:288 140520:72 +422_45x31_progressive_8bit 422 progressive 8 45 31 147851:1812 142202:4185 146387:1104 147491:288 147779:72 +422_67x45_baseline_8bit 422 baseline 8 67 45 161828:3126 149663:9045 158708:2346 161054:612 161666:162 +422_67x45_restart_8bit 422 restart 8 67 45 177119:3208 164954:9045 173999:2346 176345:612 176957:162 +422_67x45_progressive_8bit 422 progressive 8 67 45 192492:3319 180327:9045 189372:2346 191718:612 192330:162 +440_3x3_baseline_8bit 440 baseline 8 3 3 195856:371 195811:27 195838:12 195850:3 195853:3 +440_3x3_restart_8bit 440 restart 8 3 3 196272:377 196227:27 196254:12 196266:3 196269:3 +440_3x3_progressive_8bit 440 progressive 8 3 3 196694:613 196649:27 196676:12 196688:3 196691:3 +440_4x4_baseline_8bit 440 baseline 8 4 4 197373:368 197307:48 197355:12 197367:3 197370:3 +440_4x4_restart_8bit 440 restart 8 4 4 197807:374 197741:48 197789:12 197801:3 197804:3 +440_4x4_progressive_8bit 440 progressive 8 4 4 198247:605 198181:48 198229:12 198241:3 198244:3 +440_9x9_baseline_8bit 440 baseline 8 9 9 199209:448 198852:243 199095:75 199170:27 199197:12 +440_9x9_restart_8bit 440 restart 8 9 9 200014:455 199657:243 199900:75 199975:27 200002:12 +440_9x9_progressive_8bit 440 progressive 8 9 9 200826:707 200469:243 200712:75 200787:27 200814:12 +440_18x14_baseline_8bit 440 baseline 8 18 14 202556:611 201533:756 202289:189 202478:60 202538:18 +440_18x14_restart_8bit 440 restart 8 18 14 204190:622 203167:756 203923:189 204112:60 204172:18 +440_18x14_progressive_8bit 440 progressive 8 18 14 205835:869 204812:756 205568:189 205757:60 205817:18 +440_45x31_baseline_8bit 440 baseline 8 45 31 212353:1583 206704:4185 210889:1104 211993:288 212281:72 +440_45x31_restart_8bit 440 restart 8 45 31 219585:1614 213936:4185 218121:1104 219225:288 219513:72 +440_45x31_progressive_8bit 440 progressive 8 45 31 226848:1819 221199:4185 225384:1104 226488:288 226776:72 +440_67x45_baseline_8bit 440 baseline 8 67 45 240832:3079 228667:9045 237712:2346 240058:612 240670:162 +440_67x45_restart_8bit 440 restart 8 67 45 256076:3148 243911:9045 252956:2346 255302:612 255914:162 +440_67x45_progressive_8bit 440 progressive 8 67 45 271389:3252 259224:9045 268269:2346 270615:612 271227:162 +420_3x3_baseline_8bit 420 baseline 8 3 3 274686:360 274641:27 274668:12 274680:3 274683:3 +420_3x3_restart_8bit 420 restart 8 3 3 275091:366 275046:27 275073:12 275085:3 275088:3 +420_3x3_progressive_8bit 420 progressive 8 3 3 275502:599 275457:27 275484:12 275496:3 275499:3 +420_4x4_baseline_8bit 420 baseline 8 4 4 276167:363 276101:48 276149:12 276161:3 276164:3 +420_4x4_restart_8bit 420 restart 8 4 4 276596:369 276530:48 276578:12 276590:3 276593:3 +420_4x4_progressive_8bit 420 progressive 8 4 4 277031:600 276965:48 277013:12 277025:3 277028:3 +420_9x9_baseline_8bit 420 baseline 8 9 9 277988:421 277631:243 277874:75 277949:27 277976:12 +420_9x9_restart_8bit 420 restart 8 9 9 278766:427 278409:243 278652:75 278727:27 278754:12 +420_9x9_progressive_8bit 420 progressive 8 9 9 279550:669 279193:243 279436:75 279511:27 279538:12 +420_18x14_baseline_8bit 420 baseline 8 18 14 281242:550 280219:756 280975:189 281164:60 281224:18 +420_18x14_restart_8bit 420 restart 8 18 14 282815:559 281792:756 282548:189 282737:60 282797:18 +420_18x14_progressive_8bit 420 progressive 8 18 14 284397:796 283374:756 284130:189 284319:60 284379:18 +420_45x31_baseline_8bit 420 baseline 8 45 31 290842:1281 285193:4185 289378:1104 290482:288 290770:72 +420_45x31_restart_8bit 420 restart 8 45 31 297772:1299 292123:4185 296308:1104 297412:288 297700:72 +420_45x31_progressive_8bit 420 progressive 8 45 31 304720:1491 299071:4185 303256:1104 304360:288 304648:72 +420_67x45_baseline_8bit 420 baseline 8 67 45 318376:2439 306211:9045 315256:2346 317602:612 318214:162 +420_67x45_restart_8bit 420 restart 8 67 45 332980:2482 320815:9045 329860:2346 332206:612 332818:162 +420_67x45_progressive_8bit 420 progressive 8 67 45 347627:2608 335462:9045 344507:2346 346853:612 347465:162 +411_3x3_baseline_8bit 411 baseline 8 3 3 350280:352 350235:27 350262:12 350274:3 350277:3 +411_3x3_restart_8bit 411 restart 8 3 3 350677:358 350632:27 350659:12 350671:3 350674:3 +411_3x3_progressive_8bit 411 progressive 8 3 3 351080:589 351035:27 351062:12 351074:3 351077:3 +411_4x4_baseline_8bit 411 baseline 8 4 4 351735:365 351669:48 351717:12 351729:3 351732:3 +411_4x4_restart_8bit 411 restart 8 4 4 352166:371 352100:48 352148:12 352160:3 352163:3 +411_4x4_progressive_8bit 411 progressive 8 4 4 352603:600 352537:48 352585:12 352597:3 352600:3 +411_9x9_baseline_8bit 411 baseline 8 9 9 353560:462 353203:243 353446:75 353521:27 353548:12 +411_9x9_restart_8bit 411 restart 8 9 9 354379:471 354022:243 354265:75 354340:27 354367:12 +411_9x9_progressive_8bit 411 progressive 8 9 9 355207:715 354850:243 355093:75 355168:27 355195:12 +411_18x14_baseline_8bit 411 baseline 8 18 14 356945:574 355922:756 356678:189 356867:60 356927:18 +411_18x14_restart_8bit 411 restart 8 18 14 358542:582 357519:756 358275:189 358464:60 358524:18 +411_18x14_progressive_8bit 411 progressive 8 18 14 360147:818 359124:756 359880:189 360069:60 360129:18 +411_45x31_baseline_8bit 411 baseline 8 45 31 366614:1373 360965:4185 365150:1104 366254:288 366542:72 +411_45x31_restart_8bit 411 restart 8 45 31 373636:1393 367987:4185 372172:1104 373276:288 373564:72 +411_45x31_progressive_8bit 411 progressive 8 45 31 380678:1596 375029:4185 379214:1104 380318:288 380606:72 +411_67x45_baseline_8bit 411 baseline 8 67 45 394439:2488 382274:9045 391319:2346 393665:612 394277:162 +411_67x45_restart_8bit 411 restart 8 67 45 409092:2535 396927:9045 405972:2346 408318:612 408930:162 +411_67x45_progressive_8bit 411 progressive 8 67 45 423792:2676 411627:9045 420672:2346 423018:612 423630:162 +410_3x3_baseline_8bit 410 baseline 8 3 3 426513:356 426468:27 426495:12 426507:3 426510:3 +410_3x3_restart_8bit 410 restart 8 3 3 426914:362 426869:27 426896:12 426908:3 426911:3 +410_3x3_progressive_8bit 410 progressive 8 3 3 427321:593 427276:27 427303:12 427315:3 427318:3 +410_4x4_baseline_8bit 410 baseline 8 4 4 427980:365 427914:48 427962:12 427974:3 427977:3 +410_4x4_restart_8bit 410 restart 8 4 4 428411:371 428345:48 428393:12 428405:3 428408:3 +410_4x4_progressive_8bit 410 progressive 8 4 4 428848:602 428782:48 428830:12 428842:3 428845:3 +410_9x9_baseline_8bit 410 baseline 8 9 9 429807:422 429450:243 429693:75 429768:27 429795:12 +410_9x9_restart_8bit 410 restart 8 9 9 430586:428 430229:243 430472:75 430547:27 430574:12 +410_9x9_progressive_8bit 410 progressive 8 9 9 431371:671 431014:243 431257:75 431332:27 431359:12 +410_18x14_baseline_8bit 410 baseline 8 18 14 433065:524 432042:756 432798:189 432987:60 433047:18 +410_18x14_restart_8bit 410 restart 8 18 14 434612:530 433589:756 434345:189 434534:60 434594:18 +410_18x14_progressive_8bit 410 progressive 8 18 14 436165:760 435142:756 435898:189 436087:60 436147:18 +410_45x31_baseline_8bit 410 baseline 8 45 31 442574:1208 436925:4185 441110:1104 442214:288 442502:72 +410_45x31_restart_8bit 410 restart 8 45 31 449431:1223 443782:4185 447967:1104 449071:288 449359:72 +410_45x31_progressive_8bit 410 progressive 8 45 31 456303:1426 450654:4185 454839:1104 455943:288 456231:72 +410_67x45_baseline_8bit 410 baseline 8 67 45 469894:2204 457729:9045 466774:2346 469120:612 469732:162 +410_67x45_restart_8bit 410 restart 8 67 45 484263:2223 472098:9045 481143:2346 483489:612 484101:162 +410_67x45_progressive_8bit 410 progressive 8 67 45 498651:2377 486486:9045 495531:2346 497877:612 498489:162 +1x4_3x3_baseline_8bit 1x4 baseline 8 3 3 501073:345 501028:27 501055:12 501067:3 501070:3 +1x4_3x3_restart_8bit 1x4 restart 8 3 3 501463:351 501418:27 501445:12 501457:3 501460:3 +1x4_3x3_progressive_8bit 1x4 progressive 8 3 3 501859:583 501814:27 501841:12 501853:3 501856:3 +1x4_4x4_baseline_8bit 1x4 baseline 8 4 4 502508:346 502442:48 502490:12 502502:3 502505:3 +1x4_4x4_restart_8bit 1x4 restart 8 4 4 502920:352 502854:48 502902:12 502914:3 502917:3 +1x4_4x4_progressive_8bit 1x4 progressive 8 4 4 503338:580 503272:48 503320:12 503332:3 503335:3 +1x4_9x9_baseline_8bit 1x4 baseline 8 9 9 504275:441 503918:243 504161:75 504236:27 504263:12 +1x4_9x9_restart_8bit 1x4 restart 8 9 9 505073:449 504716:243 504959:75 505034:27 505061:12 +1x4_9x9_progressive_8bit 1x4 progressive 8 9 9 505879:689 505522:243 505765:75 505840:27 505867:12 +1x4_18x14_baseline_8bit 1x4 baseline 8 18 14 507591:593 506568:756 507324:189 507513:60 507573:18 +1x4_18x14_restart_8bit 1x4 restart 8 18 14 509207:604 508184:756 508940:189 509129:60 509189:18 +1x4_18x14_progressive_8bit 1x4 progressive 8 18 14 510834:840 509811:756 510567:189 510756:60 510816:18 +1x4_45x31_baseline_8bit 1x4 baseline 8 45 31 517323:1275 511674:4185 515859:1104 516963:288 517251:72 +1x4_45x31_restart_8bit 1x4 restart 8 45 31 524247:1299 518598:4185 522783:1104 523887:288 524175:72 +1x4_45x31_progressive_8bit 1x4 progressive 8 45 31 531195:1490 525546:4185 529731:1104 530835:288 531123:72 +1x4_67x45_baseline_8bit 1x4 baseline 8 67 45 544850:2597 532685:9045 541730:2346 544076:612 544688:162 +1x4_67x45_restart_8bit 1x4 restart 8 67 45 559612:2644 547447:9045 556492:2346 558838:612 559450:162 +1x4_67x45_progressive_8bit 1x4 progressive 8 67 45 574421:2757 562256:9045 571301:2346 573647:612 574259:162 +gray_4x4_extended_12bit gray extended 12 4 4 577222:234 577178:32 577210:8 577218:2 577220:2 +gray_4x4_progressive_12bit gray progressive 12 4 4 577500:357 577456:32 577488:8 577496:2 577498:2 +gray_9x9_extended_12bit gray extended 12 9 9 578095:281 577857:162 578019:50 578069:18 578087:8 +gray_9x9_progressive_12bit gray progressive 12 9 9 578614:410 578376:162 578538:50 578588:18 578606:8 +gray_45x31_extended_12bit gray extended 12 45 31 582790:1709 579024:2790 581814:736 582550:192 582742:48 +gray_45x31_progressive_12bit gray progressive 12 45 31 588265:1811 584499:2790 587289:736 588025:192 588217:48 +gray_4x64_extended_12bit gray extended 12 4 64 590764:706 590076:512 590588:128 590716:32 590748:16 +gray_4x64_progressive_12bit gray progressive 12 4 64 592158:827 591470:512 591982:128 592110:32 592142:16 +444_4x4_extended_12bit 444 extended 12 4 4 593117:483 592985:96 593081:24 593105:6 593111:6 +444_4x4_progressive_12bit 444 progressive 12 4 4 593732:730 593600:96 593696:24 593720:6 593726:6 +444_9x9_extended_12bit 444 extended 12 9 9 595176:615 594462:486 594948:150 595098:54 595152:24 +444_9x9_progressive_12bit 444 progressive 12 9 9 596505:868 595791:486 596277:150 596427:54 596481:24 +444_45x31_extended_12bit 444 extended 12 45 31 608671:4549 597373:8370 605743:2208 607951:576 608527:144 +444_45x31_progressive_12bit 444 progressive 12 45 31 624518:4798 613220:8370 621590:2208 623798:576 624374:144 +444_4x64_extended_12bit 444 extended 12 4 64 631380:1777 629316:1536 630852:384 631236:96 631332:48 +444_4x64_progressive_12bit 444 progressive 12 4 64 635221:2039 633157:1536 634693:384 635077:96 635173:48 +422_4x4_extended_12bit 422 extended 12 4 4 637392:476 637260:96 637356:24 637380:6 637386:6 +422_4x4_progressive_12bit 422 progressive 12 4 4 638000:722 637868:96 637964:24 637988:6 637994:6 +422_9x9_extended_12bit 422 extended 12 9 9 639436:570 638722:486 639208:150 639358:54 639412:24 +422_9x9_progressive_12bit 422 progressive 12 9 9 640720:822 640006:486 640492:150 640642:54 640696:24 +422_45x31_extended_12bit 422 extended 12 45 31 652840:3128 641542:8370 649912:2208 652120:576 652696:144 +422_45x31_progressive_12bit 422 progressive 12 45 31 667266:3374 655968:8370 664338:2208 666546:576 667122:144 +422_4x64_extended_12bit 422 extended 12 4 64 672704:1729 670640:1536 672176:384 672560:96 672656:48 +422_4x64_progressive_12bit 422 progressive 12 4 64 676497:1993 674433:1536 675969:384 676353:96 676449:48 +440_4x4_extended_12bit 440 extended 12 4 4 678622:478 678490:96 678586:24 678610:6 678616:6 +440_4x4_progressive_12bit 440 progressive 12 4 4 679232:719 679100:96 679196:24 679220:6 679226:6 +440_9x9_extended_12bit 440 extended 12 9 9 680665:566 679951:486 680437:150 680587:54 680641:24 +440_9x9_progressive_12bit 440 progressive 12 9 9 681945:820 681231:486 681717:150 681867:54 681921:24 +440_45x31_extended_12bit 440 extended 12 45 31 694063:3143 682765:8370 691135:2208 693343:576 693919:144 +440_45x31_progressive_12bit 440 progressive 12 45 31 708504:3392 697206:8370 705576:2208 707784:576 708360:144 +440_4x64_extended_12bit 440 extended 12 4 64 713960:1278 711896:1536 713432:384 713816:96 713912:48 +440_4x64_progressive_12bit 440 progressive 12 4 64 717302:1532 715238:1536 716774:384 717158:96 717254:48 +420_4x4_extended_12bit 420 extended 12 4 4 718966:460 718834:96 718930:24 718954:6 718960:6 +420_4x4_progressive_12bit 420 progressive 12 4 4 719558:698 719426:96 719522:24 719546:6 719552:6 +420_9x9_extended_12bit 420 extended 12 9 9 720970:527 720256:486 720742:150 720892:54 720946:24 +420_9x9_progressive_12bit 420 progressive 12 9 9 722211:780 721497:486 721983:150 722133:54 722187:24 +420_45x31_extended_12bit 420 extended 12 45 31 734289:2466 722991:8370 731361:2208 733569:576 734145:144 +420_45x31_progressive_12bit 420 progressive 12 45 31 748053:2706 736755:8370 745125:2208 747333:576 747909:144 +420_4x64_extended_12bit 420 extended 12 4 64 752823:1258 750759:1536 752295:384 752679:96 752775:48 +420_4x64_progressive_12bit 420 progressive 12 4 64 756145:1515 754081:1536 755617:384 756001:96 756097:48 +411_4x4_extended_12bit 411 extended 12 4 4 757792:471 757660:96 757756:24 757780:6 757786:6 +411_4x4_progressive_12bit 411 progressive 12 4 4 758395:708 758263:96 758359:24 758383:6 758389:6 +411_9x9_extended_12bit 411 extended 12 9 9 759817:575 759103:486 759589:150 759739:54 759793:24 +411_9x9_progressive_12bit 411 progressive 12 9 9 761106:826 760392:486 760878:150 761028:54 761082:24 +411_45x31_extended_12bit 411 extended 12 45 31 773230:2687 761932:8370 770302:2208 772510:576 773086:144 +411_45x31_progressive_12bit 411 progressive 12 45 31 787215:2923 775917:8370 784287:2208 786495:576 787071:144 +411_4x64_extended_12bit 411 extended 12 4 64 792202:1701 790138:1536 791674:384 792058:96 792154:48 +411_4x64_progressive_12bit 411 progressive 12 4 64 795967:1954 793903:1536 795439:384 795823:96 795919:48 +410_4x4_extended_12bit 410 extended 12 4 4 798053:463 797921:96 798017:24 798041:6 798047:6 +410_4x4_progressive_12bit 410 progressive 12 4 4 798648:697 798516:96 798612:24 798636:6 798642:6 +410_9x9_extended_12bit 410 extended 12 9 9 800059:531 799345:486 799831:150 799981:54 800035:24 +410_9x9_progressive_12bit 410 progressive 12 9 9 801304:781 800590:486 801076:150 801226:54 801280:24 +410_45x31_extended_12bit 410 extended 12 45 31 813383:2261 802085:8370 810455:2208 812663:576 813239:144 +410_45x31_progressive_12bit 410 progressive 12 45 31 826942:2502 815644:8370 824014:2208 826222:576 826798:144 +410_4x64_extended_12bit 410 extended 12 4 64 831508:1254 829444:1536 830980:384 831364:96 831460:48 +410_4x64_progressive_12bit 410 progressive 12 4 64 834826:1501 832762:1536 834298:384 834682:96 834778:48 +1x4_4x4_extended_12bit 1x4 extended 12 4 4 836459:394 836327:96 836423:24 836447:6 836453:6 +1x4_4x4_progressive_12bit 1x4 progressive 12 4 4 836985:625 836853:96 836949:24 836973:6 836979:6 +1x4_9x9_extended_12bit 1x4 extended 12 9 9 838324:560 837610:486 838096:150 838246:54 838300:24 +1x4_9x9_progressive_12bit 1x4 progressive 12 9 9 839598:808 838884:486 839370:150 839520:54 839574:24 +1x4_45x31_extended_12bit 1x4 extended 12 45 31 851704:2445 840406:8370 848776:2208 850984:576 851560:144 +1x4_45x31_progressive_12bit 1x4 progressive 12 45 31 865447:2690 854149:8370 862519:2208 864727:576 865303:144 +1x4_4x64_extended_12bit 1x4 extended 12 4 64 870201:1035 868137:1536 869673:384 870057:96 870153:48 +1x4_4x64_progressive_12bit 1x4 progressive 12 4 64 873300:1295 871236:1536 872772:384 873156:96 873252:48 diff --git a/crates/j2k-test-support/fixtures/scaled_matrix/data.bin b/crates/j2k-test-support/fixtures/scaled_matrix/data.bin new file mode 100644 index 000000000..05a9f51dd Binary files /dev/null and b/crates/j2k-test-support/fixtures/scaled_matrix/data.bin differ diff --git a/crates/j2k-test-support/src/fixtures.rs b/crates/j2k-test-support/src/fixtures.rs index b6a798692..abb5b8347 100644 --- a/crates/j2k-test-support/src/fixtures.rs +++ b/crates/j2k-test-support/src/fixtures.rs @@ -22,6 +22,24 @@ pub const JPEG_BASELINE_420_RESTART_32X16: &[u8] = pub const JPEG_BASELINE_420_RESTART_32X16_RGB: &[u8] = include_bytes!("../fixtures/conformance/baseline_420_restart_32x16.rgb"); +/// Synthetic 67x45 4:4:4 baseline JPEG with a 3-MCU restart interval, the +/// sampling and restart layout of Hamamatsu NDPI/VMS pyramids. The content mixes +/// gradients, flat patches (DC-only blocks), noise, and a saturated corner. +/// +/// `corpus/conformance/generate.sh` encodes it with libjpeg-turbo 3.1.4.1 +/// `cjpeg -quality 90 -sample 1x1,1x1,1x1 -baseline -optimize -restart 3B`. The +/// `_HALF`, `_QUARTER`, and `_EIGHTH` references are `djpeg -rgb -scale 1/N` +/// output from the same build (the reduced IDCTs `OpenSlide` uses for scaled +/// levels), with the PPM header stripped. +pub const JPEG_BASELINE_444_RESTART_67X45: &[u8] = + include_bytes!("../fixtures/conformance/baseline_444_restart_67x45.jpg"); +pub const JPEG_BASELINE_444_RESTART_67X45_HALF_RGB: &[u8] = + include_bytes!("../fixtures/conformance/baseline_444_restart_67x45_half.rgb"); +pub const JPEG_BASELINE_444_RESTART_67X45_QUARTER_RGB: &[u8] = + include_bytes!("../fixtures/conformance/baseline_444_restart_67x45_quarter.rgb"); +pub const JPEG_BASELINE_444_RESTART_67X45_EIGHTH_RGB: &[u8] = + include_bytes!("../fixtures/conformance/baseline_444_restart_67x45_eighth.rgb"); + /// `OpenJPEG` 2.5.4 irreversible 8x8 RGB codestream used for adapter parity tests. /// /// The source pixels are the deterministic `patterned_rgb8` formula. `OpenJPEG` diff --git a/crates/j2k-test-support/src/lib.rs b/crates/j2k-test-support/src/lib.rs index 917e4b8c2..9aac71975 100644 --- a/crates/j2k-test-support/src/lib.rs +++ b/crates/j2k-test-support/src/lib.rs @@ -18,6 +18,7 @@ mod manifest; mod metal; mod metal_shader; mod pixels; +mod scaled_matrix; pub use auto_routing::{ append_auto_routing_output, auto_routing_operation_label, auto_routing_route_cell, @@ -45,6 +46,8 @@ pub use fixtures::{ OpenJphBatchFixture, JPEG_BASELINE_420_16X16, JPEG_BASELINE_420_16X16_RGB, JPEG_BASELINE_420_RESTART_32X16, JPEG_BASELINE_420_RESTART_32X16_RGB, JPEG_BASELINE_422_16X8, JPEG_BASELINE_422_16X8_RGB, JPEG_BASELINE_444_8X8, JPEG_BASELINE_444_8X8_RGB, + JPEG_BASELINE_444_RESTART_67X45, JPEG_BASELINE_444_RESTART_67X45_EIGHTH_RGB, + JPEG_BASELINE_444_RESTART_67X45_HALF_RGB, JPEG_BASELINE_444_RESTART_67X45_QUARTER_RGB, JPEG_GRAYSCALE_8X8, JPEG_GRAYSCALE_8X8_GRAY, OPENJPEG_IRREVERSIBLE_RGB8_8X8, }; #[cfg(feature = "j2k-native-fixtures")] @@ -72,6 +75,7 @@ pub use pixels::{ project_scaled_interleaved_u8, rgb16le_to_rgba16le, rgb16ne_to_opaque_rgba16ne, rgb8_to_rgba8, scaled_rect_covering, u16_samples_to_le_bytes, PixelRect, }; +pub use scaled_matrix::{scaled_matrix_cases, ScaledMatrixCase, SCALED_MATRIX_DENOMINATORS}; /// Generates deterministic RGB8 pixels for tests and benches. pub fn patterned_rgb8(width: u32, height: u32) -> Vec { diff --git a/crates/j2k-test-support/src/scaled_matrix.rs b/crates/j2k-test-support/src/scaled_matrix.rs new file mode 100644 index 000000000..5a49d2a54 --- /dev/null +++ b/crates/j2k-test-support/src/scaled_matrix.rs @@ -0,0 +1,153 @@ +// SPDX-License-Identifier: MIT OR Apache-2.0 + +//! libjpeg-turbo DCT-scaling reference matrix. +//! +//! `corpus/conformance/scaled_matrix.py` encodes each case with libjpeg-turbo +//! 3.1.4.1 `cjpeg` and records `djpeg -dct int -scale 1/N` output for N = 1, +//! 2, 4, 8 (default fancy upsampling, as `OpenSlide` uses it). The cases cover +//! every chroma layout libjpeg-turbo decodes differently at reduced scale, +//! chroma at most two samples wide, restart intervals, progressive scans and +//! 12-bit precision. + +const CASES_TSV: &str = include_str!("../fixtures/scaled_matrix/cases.tsv"); +const DATA: &[u8] = include_bytes!("../fixtures/scaled_matrix/data.bin"); + +/// Scale denominators in reference order. +pub const SCALED_MATRIX_DENOMINATORS: [u32; 4] = [1, 2, 4, 8]; + +/// One encoded image and its libjpeg-turbo references at every scale. +#[derive(Clone, Copy, Debug)] +pub struct ScaledMatrixCase { + /// Unique case name, `{layout}_{w}x{h}_{coding}_{precision}bit`. + pub name: &'static str, + /// `gray`, `444`, `422`, `440`, `420`, `411`, `410` or `1x4` (luma + /// sampling; chroma is 1x1). + pub layout: &'static str, + /// `baseline`, `restart` (one-MCU interval), `progressive` or `extended` + /// (12-bit sequential). + pub coding: &'static str, + /// Sample precision, 8 or 12. + pub precision: u8, + /// Full-size width. + pub width: u32, + /// Full-size height. + pub height: u32, + /// The JPEG codestream. + pub jpeg: &'static [u8], + references: [&'static [u8]; 4], +} + +impl ScaledMatrixCase { + /// Whether the case is single-component grayscale. + pub fn is_gray(&self) -> bool { + self.layout == "gray" + } + + /// Interleaved samples per pixel in the references. + pub fn channels(&self) -> usize { + if self.is_gray() { + 1 + } else { + 3 + } + } + + /// Output dimensions at `1/denominator`, rounded up as libjpeg does. + pub fn scaled_dimensions(&self, denominator: u32) -> (u32, u32) { + ( + self.width.div_ceil(denominator), + self.height.div_ceil(denominator), + ) + } + + /// `djpeg -scale 1/denominator` output without its PNM header: 8-bit + /// samples, or big-endian 16-bit samples for 12-bit cases. + /// + /// # Panics + /// + /// Panics if `denominator` is not 1, 2, 4 or 8. + pub fn reference(&self, denominator: u32) -> &'static [u8] { + let index = SCALED_MATRIX_DENOMINATORS + .iter() + .position(|&d| d == denominator) + .expect("scaled matrix references cover 1/1, 1/2, 1/4 and 1/8"); + self.references[index] + } + + /// 12-bit references as sample values. + /// + /// # Panics + /// + /// Panics if `denominator` is unsupported or the case is 8-bit. + pub fn reference_u16(&self, denominator: u32) -> Vec { + assert!(self.precision > 8, "{} is an 8-bit case", self.name); + self.reference(denominator) + .chunks_exact(2) + .map(|pair| u16::from_be_bytes([pair[0], pair[1]])) + .collect() + } +} + +fn slice(field: &str) -> &'static [u8] { + let (offset, len) = field + .split_once(':') + .expect("scaled matrix slices are offset:length"); + let offset: usize = offset.parse().expect("scaled matrix offset"); + let len: usize = len.parse().expect("scaled matrix length"); + &DATA[offset..offset + len] +} + +/// Every case in the committed matrix, in generation order. +/// +/// # Panics +/// +/// Panics if the committed manifest is malformed. +pub fn scaled_matrix_cases() -> Vec { + CASES_TSV + .lines() + .filter(|line| !line.is_empty() && !line.starts_with('#')) + .map(|line| { + let fields: Vec<&'static str> = line.split('\t').collect(); + assert_eq!(fields.len(), 11, "scaled matrix row: {line}"); + ScaledMatrixCase { + name: fields[0], + layout: fields[1], + coding: fields[2], + precision: fields[3].parse().expect("precision"), + width: fields[4].parse().expect("width"), + height: fields[5].parse().expect("height"), + jpeg: slice(fields[6]), + references: [ + slice(fields[7]), + slice(fields[8]), + slice(fields[9]), + slice(fields[10]), + ], + } + }) + .collect() +} + +#[cfg(test)] +mod tests { + use super::{scaled_matrix_cases, SCALED_MATRIX_DENOMINATORS}; + + #[test] + fn every_reference_has_the_scaled_geometry() { + let cases = scaled_matrix_cases(); + assert_eq!(cases.len(), 208); + for case in cases { + assert_eq!(&case.jpeg[..2], &[0xFF, 0xD8], "{}", case.name); + let bytes_per_sample = if case.precision > 8 { 2 } else { 1 }; + for denominator in SCALED_MATRIX_DENOMINATORS { + let (w, h) = case.scaled_dimensions(denominator); + assert_eq!( + case.reference(denominator).len(), + w as usize * h as usize * case.channels() * bytes_per_sample, + "{} 1/{denominator}", + case.name + ); + } + } + } +}