Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,7 @@ categories = ["multimedia::images", "compression"]
[workspace.dependencies]
j2k-core = "=0.10.0"
j2k-metal-support = "=0.10.0"
fearless_simd = "=0.7.0"
fearless_simd = "=1.0.0"
cudarc = { version = "=0.19.9", default-features = false, features = [
"std",
"driver",
Expand Down
2 changes: 1 addition & 1 deletion THIRD_PARTY_NOTICES.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@ transcribed from ITU-T T.832. No T.835/JXRLib source code is included.
JPEG XR tables added from the ITU-T T.835 / Microsoft JPEG XR reference source
must retain the applicable BSD-3-Clause copyright and disclaimer here.

`fearless_simd` 0.7.0 is used through its safe capability-token API and is
`fearless_simd` 1.0.0 is used through its safe capability-token API and is
available under MIT or Apache-2.0 licensing.

`cudarc` 0.19.9 provides dynamically loaded CUDA Driver API and NVRTC bindings
Expand Down
3 changes: 1 addition & 2 deletions crates/jxr-core/tests/contracts.rs
Original file line number Diff line number Diff line change
Expand Up @@ -115,8 +115,7 @@ fn surface_layout_rejects_overlapping_planar_destinations() {
}

#[test]
fn core_uses_shared_backend_and_rect_contracts() {
assert_eq!(BackendKind::Cpu, BackendKind::Cpu);
fn shared_rect_accepts_contained_region() {
assert!(
Rect {
x: 1,
Expand Down
122 changes: 114 additions & 8 deletions crates/jxr-native/src/output_format/simd_pack.rs
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
//! Capability-token dispatch for common unsigned planar packing.

use fearless_simd::{Level, Simd};
use fearless_simd::{Bytes, Level, Simd, SimdBase, SimdNarrow};

use super::OutputFormatError;

Expand Down Expand Up @@ -90,28 +90,94 @@ fn pack_accelerated(
) -> bool {
#[cfg(any(target_arch = "x86", target_arch = "x86_64"))]
if let Some(avx2) = level.as_avx2() {
pack_vectorized(avx2, input, stride, start, dimensions, scaled, output);
return true;
return pack_vectorized(avx2, input, stride, start, dimensions, scaled, output);
}
#[cfg(target_arch = "aarch64")]
if let Some(neon) = level.as_neon() {
pack_vectorized(neon, input, stride, start, dimensions, scaled, output);
return true;
return pack_vectorized(neon, input, stride, start, dimensions, scaled, output);
}
false
}

#[inline]
fn pack_vectorized<S: Simd>(
_simd: S,
simd: S,
input: &[i32],
stride: usize,
start: [usize; 2],
dimensions: [usize; 2],
scaled: bool,
output: &mut [u8],
) {
pack_rows(input, stride, start, dimensions, scaled, output);
) -> bool {
let lanes = S::i32s::LEN;
let packed_lanes = S::u8s::LEN;
let [width, height] = dimensions;
if width < packed_lanes {
return false;
}

simd.vectorize(|| {
let bias = if scaled { 1_024 } else { 128 };
let rounding = if scaled { 3 } else { 0 };
let addition = bias + rounding;
let shift = if scaled { 3 } else { 0 };
let zero = S::i32s::splat(simd, 0);
let max = S::i32s::splat(simd, 255);
let vector_width = width / packed_lanes * packed_lanes;
for y in 0..height {
let source_start = (start[1] + y) * stride + start[0];
let source = &input[source_start..source_start + width];
let destination = &mut output[y * width..(y + 1) * width];
for (source, destination) in source[..vector_width]
.chunks_exact(packed_lanes)
.zip(destination[..vector_width].chunks_exact_mut(packed_lanes))
{
let first = pack_lanes(simd, &source[..lanes], addition, shift, zero, max);
let second =
pack_lanes(simd, &source[lanes..2 * lanes], addition, shift, zero, max);
let third = pack_lanes(
simd,
&source[2 * lanes..3 * lanes],
addition,
shift,
zero,
max,
);
let fourth = pack_lanes(simd, &source[3 * lanes..], addition, shift, zero, max);
first
.saturating_narrow(second)
.saturating_narrow(third.saturating_narrow(fourth))
.store_slice(destination);
}
for (destination, &sample) in destination[vector_width..]
.iter_mut()
.zip(&source[vector_width..])
{
*destination = u8::try_from(((sample + bias + rounding) >> shift).clamp(0, 255))
.expect("sample is clipped to u8");
}
}
});
true
}

#[expect(
clippy::inline_always,
reason = "SIMD helper must inline into the target-feature vectorize context"
)]
#[inline(always)]
fn pack_lanes<S: Simd>(
simd: S,
source: &[i32],
addition: i32,
shift: u32,
zero: S::i32s,
max: S::i32s,
) -> S::u32s {
((S::i32s::from_slice(simd, source) + addition) >> shift)
.max(zero)
.min(max)
.bitcast()
}

#[inline]
Expand Down Expand Up @@ -183,4 +249,44 @@ mod tests {
assert_eq!(output, expected);
}
}

#[test]
fn packing_matches_scalar_at_row_tails_and_clamp_boundaries() {
let width = 37;
let height = 3;
let stride = 41;
let input: Vec<i32> = (0..stride * (height + 1))
.map(|index| match index % 5 {
0 => -2_000,
1 => -128,
2 => 0,
3 => 1_024,
_ => i32::MAX - 1_027,
})
.collect();
for scaled in [false, true] {
let mut output = Vec::new();
append_u8(
Level::new(),
&input,
stride,
[2, 1],
[width, height],
scaled,
&mut output,
)
.unwrap();
let bias = if scaled { 1_024 } else { 128 };
let rounding = if scaled { 3 } else { 0 };
let shift = if scaled { 3 } else { 0 };
let expected: Vec<_> = (1..=height)
.flat_map(|y| &input[y * stride + 2..y * stride + 2 + width])
.map(|&sample| {
u8::try_from(((sample + bias + rounding) >> shift).clamp(0, 255))
.expect("sample is clipped to u8")
})
.collect();
assert_eq!(output, expected);
}
}
}
35 changes: 32 additions & 3 deletions crates/jxr-native/src/reconstruct/simd_dequant.rs
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
//! Checked capability-token SIMD for contiguous coefficient scaling.

use fearless_simd::Simd;
use fearless_simd::{Simd, SimdBase};
use jxr_math::quantization::Quantizer;

use crate::CpuCapabilities;
Expand Down Expand Up @@ -49,8 +49,24 @@ pub(super) fn scale_coefficients(
}

#[inline]
fn scale_vectorized<S: Simd>(_simd: S, input: &[i32], output: &mut [i32], step: u32) {
scale_validated(input, output, step);
fn scale_vectorized<S: Simd>(simd: S, input: &[i32], output: &mut [i32], step: u32) {
let multiplier = i32::from_ne_bytes(step.to_ne_bytes());
simd.vectorize(|| {
let lanes = S::i32s::LEN;
let vector_multiplier = S::i32s::splat(simd, multiplier);
let mut input_chunks = input.chunks_exact(lanes);
let mut output_chunks = output.chunks_exact_mut(lanes);
for (source, destination) in input_chunks.by_ref().zip(output_chunks.by_ref()) {
(S::i32s::from_slice(simd, source) * vector_multiplier).store_slice(destination);
}
for (&source, destination) in input_chunks
.remainder()
.iter()
.zip(output_chunks.into_remainder())
{
*destination = source.wrapping_mul(multiplier);
}
});
}

#[inline]
Expand Down Expand Up @@ -99,4 +115,17 @@ mod tests {
.is_err()
);
}

#[test]
fn coefficient_scaling_preserves_negative_values_and_tail() {
let input: Vec<_> = (-34..35).collect();
let quantizer = Quantizer::new(113).unwrap();
let mut output = vec![0; input.len()];
scale_coefficients(CpuCapabilities::detect(), quantizer, &input, &mut output).unwrap();
let expected: Vec<_> = input
.iter()
.map(|&value| quantizer.dequantize(value).unwrap())
.collect();
assert_eq!(output, expected);
}
}
4 changes: 2 additions & 2 deletions crates/jxr/fuzz/Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading