Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 27 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,33 @@ and stale roadmap entries have been removed from the public documentation set.
- Improves grouped JPEG Metal decode, pooled output reuse, HT cleanup,
and inverse wavelet transform dispatch with regression coverage.

- JPEG DCT-scaled decodes (1/2, 1/4, 1/8) now match libjpeg-turbo, and
therefore OpenSlide's NDPI/VMS levels, bit for bit. Checked against a
committed libjpeg-turbo 3.1.4.1 reference matrix covering every chroma
layout, restart intervals, progressive scans and 12-bit precision:
- Reduced IDCTs round like libjpeg's `DESCALE` instead of truncating.
- Each component gets libjpeg-turbo's IDCT size: 4:2:0 and 4:1:0 chroma is
decoded with a larger reduced IDCT, so 4:2:0 needs no upsampling below
full size.
- Chroma is replicated instead of smoothed at 1/8 scale, and at any scale
(including full size) when 2:1 chroma is at most two samples wide.
- Progressive images apply the reduced IDCT to their coefficients instead
of decimating a full-size decode.
- 12-bit images use 12-bit reduced IDCTs instead of decimating a full-size
decode. 12-bit 4:4:0, 4:1:1, 4:1:0 and 1x4 layouts still return
`NotImplemented`.
- Fixes a panic in 12-bit region decodes when the region excluded image
blocks on its left or right.
- Includes output rows in the 12-bit scratch budget so tall, narrow JPEGs
decode without false memory-cap failures.
- `j2k-jpeg-cuda` CPU-backed scaled and region-scaled surfaces now have the
scaled dimensions; they were sized from the source rectangle.
- JPEG Metal scaled decodes follow the same rules, and 4:2:0/4:2:2 region
decodes keep the neighbouring chroma their smoothing reads when a region
ends inside the image. JPEG images at most four pixels wide with 4:2:0 or
4:2:2 chroma are no longer Metal or CUDA fast shapes: explicit Metal
requests reject them and automatic routing decodes them on the CPU.

- Upgrades `fearless_simd` to 1.0.0 for the `j2k-jpeg` AArch64 NEON kernels
and the optional `j2k-native` `simd` feature. The dependency is private, so
public APIs are unchanged. `j2k-jpeg` keeps its private exact-AVX2 token on
Expand Down
Binary file added corpus/conformance/baseline_444_restart_67x45.jpg
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
Binary file not shown.
Binary file not shown.
Binary file not shown.
99 changes: 99 additions & 0 deletions corpus/conformance/generate.sh
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,48 @@ sys.stdout.buffer.write(header + body)
PY
}

# Integer-only texture (no floating point or library RNG) so every host
# produces the same pixels: triangle waves, two flat patches that code as
# DC-only blocks, LCG noise, and a saturated corner.
write_texture_ppm() {
local width="$1"
local height="$2"
python3 - "$width" "$height" <<'PY'
import sys

width = int(sys.argv[1])
height = int(sys.argv[2])
state = 20260927


def noise():
global state
state = (state * 1103515245 + 12345) & 0x7FFFFFFF
return state % 121 - 60


header = f"P6\n{width} {height}\n255\n".encode()
body = bytearray()
for y in range(height):
for x in range(width):
rgb = [
28 + abs((x * 9 + y * 4) % 144 - 72) * 200 // 72,
40 + x * 180 // max(width - 1, 1),
200 - y * 150 // max(height - 1, 1) + abs((x + y) * 7 % 64 - 32) - 16,
]
if y < 16 and 40 <= x < 56:
rgb = [201, 77, 143]
if 24 <= y < 40 and 8 <= x < 24:
rgb = [19, 233, 98]
if y >= 30 and x >= 40:
rgb = [value + noise() for value in rgb]
if y < 10 and x < 12:
rgb = [255, 255, 0]
body.extend(min(max(value, 0), 255) for value in rgb)
sys.stdout.buffer.write(header + bytes(body))
PY
}

strip_pnm_header() {
python3 -c 'import sys
data = sys.stdin.buffer.read()
Expand Down Expand Up @@ -113,6 +155,20 @@ write_gray_pgm 8 8 \
djpeg -grayscale grayscale_8x8.jpg | strip_pnm_header > grayscale_8x8.gray
assert_size grayscale_8x8.gray 64

# DCT-scaled references: the reduced IDCTs behind OpenSlide's NDPI/VMS levels.
write_texture_ppm 67 45 \
| cjpeg -quality 90 -sample 1x1,1x1,1x1 -baseline -optimize -restart 3B \
-outfile baseline_444_restart_67x45.jpg
djpeg -rgb -scale 1/2 baseline_444_restart_67x45.jpg \
| strip_pnm_header > baseline_444_restart_67x45_half.rgb
assert_size baseline_444_restart_67x45_half.rgb 2346
djpeg -rgb -scale 1/4 baseline_444_restart_67x45.jpg \
| strip_pnm_header > baseline_444_restart_67x45_quarter.rgb
assert_size baseline_444_restart_67x45_quarter.rgb 612
djpeg -rgb -scale 1/8 baseline_444_restart_67x45.jpg \
| strip_pnm_header > baseline_444_restart_67x45_eighth.rgb
assert_size baseline_444_restart_67x45_eighth.rgb 162

cat > manifest.json <<EOF
{
"libjpeg_turbo_version": "$LJT_VERSION",
Expand Down Expand Up @@ -172,10 +228,53 @@ cat > manifest.json <<EOF
"height": 8,
"tolerance": "bit_exact",
"sampling": "grayscale"
},
{
"input": "baseline_444_restart_67x45.jpg",
"input_sha256": "$(sha256_file baseline_444_restart_67x45.jpg)",
"reference": "baseline_444_restart_67x45_half.rgb",
"reference_sha256": "$(sha256_file baseline_444_restart_67x45_half.rgb)",
"format": "Rgb8",
"width": 34,
"height": 23,
"scale": "1/2",
"tolerance": "bit_exact",
"sampling": "4:4:4 restart-coded",
"generator": "$LJT_VERSION"
},
{
"input": "baseline_444_restart_67x45.jpg",
"input_sha256": "$(sha256_file baseline_444_restart_67x45.jpg)",
"reference": "baseline_444_restart_67x45_quarter.rgb",
"reference_sha256": "$(sha256_file baseline_444_restart_67x45_quarter.rgb)",
"format": "Rgb8",
"width": 17,
"height": 12,
"scale": "1/4",
"tolerance": "bit_exact",
"sampling": "4:4:4 restart-coded",
"generator": "$LJT_VERSION"
},
{
"input": "baseline_444_restart_67x45.jpg",
"input_sha256": "$(sha256_file baseline_444_restart_67x45.jpg)",
"reference": "baseline_444_restart_67x45_eighth.rgb",
"reference_sha256": "$(sha256_file baseline_444_restart_67x45_eighth.rgb)",
"format": "Rgb8",
"width": 9,
"height": 6,
"scale": "1/8",
"tolerance": "bit_exact",
"sampling": "4:4:4 restart-coded",
"generator": "$LJT_VERSION"
}
]
}
EOF

# DCT-scaling reference matrix (every chroma layout, progressive and 12-bit)
# for the scaled-decode tests; written to j2k-test-support/fixtures.
python3 scaled_matrix.py

echo "Regenerated fixtures:"
ls -la *.jpg *.rgb *.gray manifest.json
39 changes: 39 additions & 0 deletions corpus/conformance/manifest.json
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,45 @@
"height": 8,
"tolerance": "bit_exact",
"sampling": "grayscale"
},
{
"input": "baseline_444_restart_67x45.jpg",
"input_sha256": "990f70470210e6fe4e2ccd62ba2a936d52bd34557971083fc46860351acab097",
"reference": "baseline_444_restart_67x45_half.rgb",
"reference_sha256": "df178855cf57967e10210725ea8dd00a8b1000033c685ee48604f03b72b6cb16",
"format": "Rgb8",
"width": 34,
"height": 23,
"scale": "1/2",
"tolerance": "bit_exact",
"sampling": "4:4:4 restart-coded",
"generator": "libjpeg-turbo version 3.1.4.1 (build 20260327)"
},
{
"input": "baseline_444_restart_67x45.jpg",
"input_sha256": "990f70470210e6fe4e2ccd62ba2a936d52bd34557971083fc46860351acab097",
"reference": "baseline_444_restart_67x45_quarter.rgb",
"reference_sha256": "ae115fa11103c93f7c53d7a5b2c03b495319b229459d4cad9097bac038dee27d",
"format": "Rgb8",
"width": 17,
"height": 12,
"scale": "1/4",
"tolerance": "bit_exact",
"sampling": "4:4:4 restart-coded",
"generator": "libjpeg-turbo version 3.1.4.1 (build 20260327)"
},
{
"input": "baseline_444_restart_67x45.jpg",
"input_sha256": "990f70470210e6fe4e2ccd62ba2a936d52bd34557971083fc46860351acab097",
"reference": "baseline_444_restart_67x45_eighth.rgb",
"reference_sha256": "842c61a0aa8483ce96acc7cf44058b5f7967df39ec2b39cda77f6b12840d029d",
"format": "Rgb8",
"width": 9,
"height": 6,
"scale": "1/8",
"tolerance": "bit_exact",
"sampling": "4:4:4 restart-coded",
"generator": "libjpeg-turbo version 3.1.4.1 (build 20260327)"
}
]
}
183 changes: 183 additions & 0 deletions corpus/conformance/scaled_matrix.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,183 @@
#!/usr/bin/env python3
# SPDX-License-Identifier: MIT OR Apache-2.0
"""Regenerate the libjpeg-turbo DCT-scaling reference matrix.

Run manually (it is also invoked by generate.sh) and commit the output. CI does
not run it; tests read the committed files.

Every case is encoded with `cjpeg` and decoded with `djpeg -scale 1/N` for
N = 1, 2, 4, 8 using default (fancy) upsampling and the accurate integer IDCT,
which is how OpenSlide scale-decodes NDPI/VMS levels. The matrix covers the
libjpeg-turbo behaviours j2k must reproduce:

- chroma that libjpeg-turbo decodes with a larger reduced IDCT so it needs less
or no upsampling (4:2:0, 4:1:0 at 1/2, 1/4, 1/8);
- box (replicating) upsampling at 1/8, where libjpeg-turbo disables smoothing;
- box upsampling when a smoothed 2:1 component is at most 2 samples wide;
- progressive and 12-bit reduced decodes.

Output (in crates/j2k-test-support/fixtures/scaled_matrix/):

- cases.tsv one line per case: name, layout, coding, precision, width,
height, then offset:length of the JPEG and of each reference in
data.bin (full, 1/2, 1/4, 1/8).
- data.bin the JPEG inputs and headerless references, concatenated. 8-bit
references are interleaved RGB or gray bytes; 12-bit references
are big-endian 16-bit samples as djpeg writes them.

Requires libjpeg-turbo 3.x `cjpeg` and `djpeg` on PATH.
"""

import os
import subprocess
import sys

HERE = os.path.dirname(os.path.abspath(__file__))
OUT_DIR = os.path.join(
HERE, "..", "..", "crates", "j2k-test-support", "fixtures", "scaled_matrix"
)

# name -> cjpeg -sample argument (luma factors; chroma are 1x1), or None for
# grayscale.
LAYOUTS = [
("gray", None),
("444", "1x1"),
("422", "2x1"),
("440", "1x2"),
("420", "2x2"),
("411", "4x1"),
("410", "4x2"),
("1x4", "1x4"),
]

# 3 and 4 pixel wide images leave 2:1 chroma at most 2 samples wide at full
# size; 9 does at 1/4; the larger sizes span several MCUs and partial edges.
SIZES_8BIT = [(3, 3), (4, 4), (9, 9), (18, 14), (45, 31), (67, 45)]
# Tall narrow images make the component planes dominate the scratch budget,
# while full-size 2:1 chroma still takes the scaled (replicating) writer.
SIZES_12BIT = [(4, 4), (9, 9), (45, 31), (4, 64)]

CODINGS_8BIT = [
("baseline", ["-baseline", "-optimize"]),
("restart", ["-baseline", "-optimize", "-restart", "1B"]),
("progressive", ["-progressive"]),
]
CODINGS_12BIT = [
("extended", ["-optimize"]),
("progressive", ["-progressive"]),
]

SCALES = [1, 2, 4, 8]


def texture(width, height, precision):
"""Deterministic integer-only texture with strong chroma detail."""
state = 20260927
maxval = (1 << precision) - 1
shift = precision - 8
pixels = []
for y in range(height):
for x in range(width):
rgb = [
28 + abs((x * 9 + y * 4) % 144 - 72) * 200 // 72,
40 + x * 180 // max(width - 1, 1),
200 - y * 150 // max(height - 1, 1) + abs((x + y) * 7 % 64 - 32) - 16,
]
if (x + 2 * y) % 5 == 0:
rgb = [rgb[2], rgb[0], rgb[1]]
if y >= height // 2 and x >= width // 2:
state = (state * 1103515245 + 12345) & 0x7FFFFFFF
rgb = [value + state % 121 - 60 for value in rgb]
# Saturated corner, kept off tiny images so they stay textured.
if width >= 9 and y < 3 and x < 3:
rgb = [255, 255, 0]
rgb = [min(max(value, 0), 255) for value in rgb]
if shift:
rgb = [(value << shift) | ((x * 5 + y * 3 + i) % (1 << shift)) for i, value in enumerate(rgb)]
rgb = [min(value, maxval) for value in rgb]
pixels.append(rgb)
return pixels


def pnm(width, height, precision, gray):
maxval = (1 << precision) - 1
magic = "P5" if gray else "P6"
body = bytearray()
for rgb in texture(width, height, precision):
samples = [(rgb[0] * 77 + rgb[1] * 150 + rgb[2] * 29) >> 8] if gray else rgb
for value in samples:
if maxval > 255:
body += value.to_bytes(2, "big")
else:
body.append(value)
return f"{magic}\n{width} {height}\n{maxval}\n".encode() + bytes(body)


def strip_pnm_header(data):
index = 0
for _ in range(3):
index = data.index(b"\n", index) + 1
return data[index:]


def run(args, stdin):
return subprocess.run(args, input=stdin, capture_output=True, check=True).stdout


def main():
probe = subprocess.run(["cjpeg", "-version"], capture_output=True, check=False)
version = (probe.stderr or probe.stdout).decode().strip().splitlines()[0]
if "libjpeg-turbo version 3." not in version:
sys.exit(f"error: need libjpeg-turbo 3.x cjpeg/djpeg, found {version!r}")

os.makedirs(OUT_DIR, exist_ok=True)
data = bytearray()
lines = [
f"# Generated by corpus/conformance/scaled_matrix.py with {version}",
"# name\tlayout\tcoding\tprecision\twidth\theight\tjpeg\tscale1\tscale2\tscale4\tscale8",
]

def append(blob):
start = len(data)
data.extend(blob)
return f"{start}:{len(blob)}"

matrix = [(8, SIZES_8BIT, CODINGS_8BIT), (12, SIZES_12BIT, CODINGS_12BIT)]
for precision, sizes, codings in matrix:
for layout, sample in LAYOUTS:
gray = sample is None
for width, height in sizes:
for coding, coding_args in codings:
args = ["cjpeg", "-quality", "90", "-precision", str(precision)]
args += ["-grayscale"] if gray else ["-sample", f"{sample},1x1,1x1"]
jpeg = run(args + coding_args, pnm(width, height, precision, gray))
refs = []
for denom in SCALES:
out = ["djpeg", "-dct", "int", "-scale", f"1/{denom}"]
out += ["-grayscale"] if gray else ["-rgb"]
ref = strip_pnm_header(run(out, jpeg))
channels = 1 if gray else 3
bytes_per_sample = 2 if precision > 8 else 1
expected = (
-(-width // denom) * -(-height // denom) * channels * bytes_per_sample
)
if len(ref) != expected:
sys.exit(f"error: {layout} {width}x{height} {coding} 1/{denom}: {len(ref)} != {expected}")
refs.append(append(ref))
name = f"{layout}_{width}x{height}_{coding}_{precision}bit"
lines.append(
"\t".join(
[name, layout, coding, str(precision), str(width), str(height), append(jpeg)]
+ refs
)
)

with open(os.path.join(OUT_DIR, "data.bin"), "wb") as handle:
handle.write(data)
with open(os.path.join(OUT_DIR, "cases.tsv"), "w", encoding="utf-8") as handle:
handle.write("\n".join(lines) + "\n")
print(f"wrote {len(lines) - 2} cases, {len(data)} bytes")


if __name__ == "__main__":
main()
Loading
Loading