/rust/registry/src/index.crates.io-1949cf8c6b5b557f/zune-jpeg-0.5.15/src/mcu.rs
Line | Count | Source |
1 | | /* |
2 | | * Copyright (c) 2023. |
3 | | * |
4 | | * This software is free software; |
5 | | * |
6 | | * You can redistribute it or modify it under terms of the MIT, Apache License or Zlib license |
7 | | */ |
8 | | |
9 | | use alloc::vec::Vec; |
10 | | use alloc::{format, vec}; |
11 | | use core::cmp::min; |
12 | | |
13 | | use zune_core::bytestream::ZByteReaderTrait; |
14 | | use zune_core::colorspace::ColorSpace; |
15 | | use zune_core::colorspace::ColorSpace::Luma; |
16 | | use zune_core::log::{error, trace, warn}; |
17 | | |
18 | | use crate::bitstream::BitStream; |
19 | | use crate::components::SampleRatios; |
20 | | use crate::decoder::MAX_COMPONENTS; |
21 | | use crate::errors::DecodeErrors; |
22 | | use crate::marker::Marker; |
23 | | use crate::mcu_prog::get_marker; |
24 | | use crate::misc::{calculate_padded_width, setup_component_params}; |
25 | | use crate::worker::{color_convert, upsample}; |
26 | | use crate::JpegDecoder; |
27 | | |
28 | | /// The size of a DC block for a MCU. |
29 | | |
30 | | pub const DCT_BLOCK: usize = 64; |
31 | | |
32 | | impl<T: ZByteReaderTrait> JpegDecoder<T> { |
33 | | /// Check for existence of DC and AC Huffman Tables |
34 | 1.03k | pub(crate) fn check_tables(&self) -> Result<(), DecodeErrors> { |
35 | | // check that dc and AC tables exist outside the hot path |
36 | 4.47k | for component in &self.components { |
37 | 3.44k | let _ = &self |
38 | 3.44k | .dc_huffman_tables |
39 | 3.44k | .get(component.dc_huff_table) |
40 | 3.44k | .as_ref() |
41 | 3.44k | .ok_or_else(|| { |
42 | 1 | DecodeErrors::HuffmanDecode(format!( |
43 | 1 | "No Huffman DC table for component {:?} ", |
44 | 1 | component.component_id |
45 | 1 | )) |
46 | 1 | })? Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::check_tables::{closure#0}<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::check_tables::{closure#0}Line | Count | Source | 41 | 1 | .ok_or_else(|| { | 42 | 1 | DecodeErrors::HuffmanDecode(format!( | 43 | 1 | "No Huffman DC table for component {:?} ", | 44 | 1 | component.component_id | 45 | 1 | )) | 46 | 1 | })? |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::check_tables::{closure#0} |
47 | 3.44k | .as_ref() |
48 | 3.44k | .ok_or_else(|| { |
49 | 0 | DecodeErrors::HuffmanDecode(format!( |
50 | 0 | "No DC table for component {:?}", |
51 | 0 | component.component_id |
52 | 0 | )) |
53 | 0 | })?; Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::check_tables::{closure#1}Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::check_tables::{closure#1}Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::check_tables::{closure#1} |
54 | | |
55 | 3.44k | let _ = &self |
56 | 3.44k | .ac_huffman_tables |
57 | 3.44k | .get(component.ac_huff_table) |
58 | 3.44k | .as_ref() |
59 | 3.44k | .ok_or_else(|| { |
60 | 0 | DecodeErrors::HuffmanDecode(format!( |
61 | 0 | "No Huffman AC table for component {:?} ", |
62 | 0 | component.component_id |
63 | 0 | )) |
64 | 0 | })? Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::check_tables::{closure#2}Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::check_tables::{closure#2}Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::check_tables::{closure#2} |
65 | 3.44k | .as_ref() |
66 | 3.44k | .ok_or_else(|| { |
67 | 0 | DecodeErrors::HuffmanDecode(format!( |
68 | 0 | "No AC table for component {:?}", |
69 | 0 | component.component_id |
70 | 0 | )) |
71 | 0 | })?; Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::check_tables::{closure#3}Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::check_tables::{closure#3}Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::check_tables::{closure#3} |
72 | | } |
73 | 1.03k | Ok(()) |
74 | 1.03k | } Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::check_tables <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::check_tables Line | Count | Source | 34 | 1.03k | pub(crate) fn check_tables(&self) -> Result<(), DecodeErrors> { | 35 | | // check that dc and AC tables exist outside the hot path | 36 | 4.47k | for component in &self.components { | 37 | 3.44k | let _ = &self | 38 | 3.44k | .dc_huffman_tables | 39 | 3.44k | .get(component.dc_huff_table) | 40 | 3.44k | .as_ref() | 41 | 3.44k | .ok_or_else(|| { | 42 | | DecodeErrors::HuffmanDecode(format!( | 43 | | "No Huffman DC table for component {:?} ", | 44 | | component.component_id | 45 | | )) | 46 | 1 | })? | 47 | 3.44k | .as_ref() | 48 | 3.44k | .ok_or_else(|| { | 49 | | DecodeErrors::HuffmanDecode(format!( | 50 | | "No DC table for component {:?}", | 51 | | component.component_id | 52 | | )) | 53 | 0 | })?; | 54 | | | 55 | 3.44k | let _ = &self | 56 | 3.44k | .ac_huffman_tables | 57 | 3.44k | .get(component.ac_huff_table) | 58 | 3.44k | .as_ref() | 59 | 3.44k | .ok_or_else(|| { | 60 | | DecodeErrors::HuffmanDecode(format!( | 61 | | "No Huffman AC table for component {:?} ", | 62 | | component.component_id | 63 | | )) | 64 | 0 | })? | 65 | 3.44k | .as_ref() | 66 | 3.44k | .ok_or_else(|| { | 67 | | DecodeErrors::HuffmanDecode(format!( | 68 | | "No AC table for component {:?}", | 69 | | component.component_id | 70 | | )) | 71 | 0 | })?; | 72 | | } | 73 | 1.03k | Ok(()) | 74 | 1.03k | } |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::check_tables |
75 | | |
76 | | /// Decode MCUs and carry out post processing. |
77 | | /// |
78 | | /// This is the main decoder loop for the library, the hot path. |
79 | | /// |
80 | | /// Because of this, we pull in some very crazy optimization tricks hence readability is a pinch |
81 | | /// here. |
82 | | #[allow( |
83 | | clippy::similar_names, |
84 | | clippy::too_many_lines, |
85 | | clippy::cast_possible_truncation |
86 | | )] |
87 | | #[inline(never)] |
88 | 1.03k | pub(crate) fn decode_mcu_ycbcr_baseline( |
89 | 1.03k | &mut self, pixels: &mut [u8] |
90 | 1.03k | ) -> Result<(), DecodeErrors> { |
91 | 1.03k | setup_component_params(self)?; |
92 | | |
93 | | // check dc and AC tables |
94 | 1.03k | self.check_tables()?; |
95 | | |
96 | | let (mut mcu_width, mut mcu_height); |
97 | | |
98 | 1.03k | if self.is_interleaved { |
99 | | // set upsampling functions |
100 | 584 | self.set_upsampling()?; |
101 | | |
102 | 584 | mcu_width = self.mcu_x; |
103 | 584 | mcu_height = self.mcu_y; |
104 | 447 | } else { |
105 | 447 | // For non-interleaved images( (1*1) subsampling) |
106 | 447 | // number of MCU's are the widths (+7 to account for paddings) divided bu 8. |
107 | 447 | mcu_width = (self.info.width as usize + 7) / 8; |
108 | 447 | mcu_height = (self.info.height as usize + 7) / 8; |
109 | 447 | } |
110 | 1.03k | if self.is_interleaved |
111 | 584 | && self.input_colorspace.num_components() > 1 |
112 | 577 | && self.options.jpeg_get_out_colorspace().num_components() == 1 |
113 | 0 | && (self.info.sample_ratio == SampleRatios::V |
114 | 0 | || self.info.sample_ratio == SampleRatios::HV) |
115 | 0 | { |
116 | 0 | // For a specific set of images, e.g interleaved, |
117 | 0 | // when converting from YcbCr to grayscale, we need to |
118 | 0 | // take into account mcu height since the MCU decoding needs to take |
119 | 0 | // it into account for padding purposes and the post processor |
120 | 0 | // parses two rows per mcu width. |
121 | 0 | // |
122 | 0 | // set coeff to be 2 to ensure that we increment two rows |
123 | 0 | // for every mcu processed also |
124 | 0 | mcu_height *= self.v_max; |
125 | 0 | mcu_height /= self.h_max; |
126 | 0 | self.coeff = 2; |
127 | 1.03k | } |
128 | | |
129 | 1.03k | if self.input_colorspace == ColorSpace::Luma && self.is_interleaved { |
130 | 6 | warn!("Grayscale image with down-sampled component, resetting component details"); |
131 | 6 | |
132 | 6 | self.reset_params(); |
133 | 6 | |
134 | 6 | mcu_width = ((self.info.width + 7) / 8) as usize; |
135 | 6 | mcu_height = ((self.info.height + 7) / 8) as usize; |
136 | 1.02k | } |
137 | 1.03k | let width = usize::from(self.info.width); |
138 | | |
139 | 1.03k | let padded_width = calculate_padded_width(width, self.info.sample_ratio); |
140 | | |
141 | 1.03k | let mut stream = BitStream::new(); |
142 | 1.03k | let mut tmp = [0_i32; DCT_BLOCK]; |
143 | | |
144 | 1.03k | let comp_len = self.components.len(); |
145 | | |
146 | 3.44k | for (pos, comp) in self.components.iter_mut().enumerate() { |
147 | | // Allocate only needed components. |
148 | | // |
149 | | // For special colorspaces i.e YCCK and CMYK, just allocate all of the needed |
150 | | // components. |
151 | 3.44k | if min( |
152 | 3.44k | self.options.jpeg_get_out_colorspace().num_components() - 1, |
153 | 3.44k | pos |
154 | 3.44k | ) == pos |
155 | 2 | || comp_len == 4 |
156 | | // Special colorspace |
157 | 3.44k | { |
158 | 3.44k | // allocate enough space to hold a whole MCU width |
159 | 3.44k | // this means we should take into account sampling ratios |
160 | 3.44k | // `*8` is because each MCU spans 8 widths. |
161 | 3.44k | let len = comp.width_stride * comp.vertical_sample * 8; |
162 | 3.44k | |
163 | 3.44k | comp.needed = true; |
164 | 3.44k | comp.raw_coeff = vec![0; len]; |
165 | 3.44k | } else { |
166 | 0 | comp.needed = false; |
167 | 0 | } |
168 | | } |
169 | | |
170 | | // If all components are contained in the first scan of MCUs, then we can process into |
171 | | // (upsampled) pixels immediately after each MCU, for convenience we use each row of MCUS. |
172 | | // Otherwise, we must first wait until following SOS provide the remaining components. |
173 | 1.03k | let all_components_in_first_scan = usize::from(self.num_scans) == self.components.len(); |
174 | 4.12k | let mut progressive_mcus: [Vec<i16>; 4] = core::array::from_fn(|_| vec![]); Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::decode_mcu_ycbcr_baseline::{closure#0}<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::decode_mcu_ycbcr_baseline::{closure#0}Line | Count | Source | 174 | 4.12k | let mut progressive_mcus: [Vec<i16>; 4] = core::array::from_fn(|_| vec![]); |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::decode_mcu_ycbcr_baseline::{closure#0} |
175 | | |
176 | 1.03k | if !all_components_in_first_scan { |
177 | 780 | for (component, mcu) in self.components.iter().zip(&mut progressive_mcus) { |
178 | 780 | let len = mcu_width |
179 | 780 | * component.vertical_sample |
180 | 780 | * component.horizontal_sample |
181 | 780 | * mcu_height |
182 | 780 | * 64; |
183 | 780 | *mcu = vec![0; len]; |
184 | 780 | } |
185 | 836 | } |
186 | | |
187 | 1.03k | let mut pixels_written = 0; |
188 | | |
189 | 1.03k | let is_hv = usize::from(self.is_interleaved); |
190 | 1.03k | let upsampler_scratch_size = is_hv * self.components.iter().map(|x| x.width_stride).max().unwrap_or(0) * 8; |
191 | 1.03k | let mut upsampler_scratch_space = vec![0; upsampler_scratch_size]; |
192 | | |
193 | | 'sos: loop { |
194 | | trace!( |
195 | | "Baseline decoding of components: {:?}", |
196 | | &self.z_order[..usize::from(self.num_scans)] |
197 | | ); |
198 | | |
199 | | trace!("Decoding MCU width: {mcu_width}, height: {mcu_height}"); |
200 | | |
201 | 38.2k | for i in 0..mcu_height { |
202 | 38.2k | if stream.overread_by > 0 { |
203 | 42 | pixels.get_mut(pixels_written..).map(|v| v.fill(128)); Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::decode_mcu_ycbcr_baseline::{closure#2}<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::decode_mcu_ycbcr_baseline::{closure#2}Line | Count | Source | 203 | 42 | pixels.get_mut(pixels_written..).map(|v| v.fill(128)); |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::decode_mcu_ycbcr_baseline::{closure#2} |
204 | 42 | if self.options.strict_mode() { |
205 | 0 | return Err(DecodeErrors::FormatStatic("Premature end of buffer")); |
206 | 42 | }; |
207 | | |
208 | 42 | error!("Premature end of buffer"); |
209 | 42 | break; |
210 | 38.2k | } |
211 | | |
212 | | // decode a whole MCU width, |
213 | | // this takes into account interleaved components. |
214 | 38.2k | let terminate = if all_components_in_first_scan { |
215 | 30.8k | self.decode_mcu_width::<false>( |
216 | 30.8k | mcu_width, |
217 | 30.8k | i, |
218 | 30.8k | &mut tmp, |
219 | 30.8k | &mut stream, |
220 | 30.8k | &mut progressive_mcus |
221 | 9 | )? |
222 | | } else { |
223 | | /* NB: (cae). This code was added due to the issue at https://github.com/etemesi254/zune-image/issues/277 |
224 | | * |
225 | | * There is a particular set of images that interleave the start of scan (SOS) with the MCU, |
226 | | * E.g if it's a three component image, we have SOS->MCU ->SOS->MCU ->SOS->MCU |
227 | | * which presents a problem on decoding, we need to buffer the whole image before continuing since |
228 | | * we won't have a row containing all the component data which will be needed e.g for color conversion. |
229 | | * |
230 | | * The mechanisms is that we decode the whole image upfront, which goes against the normal |
231 | | * routine of decoding MCU width , so this requires more memory upfront than initial routines |
232 | | * but it is a single image out of the many corpuses that exist, so its fine. |
233 | | * (image in test-images/jpeg/sos_news.jpeg) |
234 | | |
235 | | * Code contributed by Aurelia Molzer (https://github.com/197g) |
236 | | |
237 | | * |
238 | | */ |
239 | | |
240 | 7.36k | self.decode_mcu_width::<true>( |
241 | 7.36k | mcu_width, |
242 | 7.36k | i, |
243 | 7.36k | &mut tmp, |
244 | 7.36k | &mut stream, |
245 | 7.36k | &mut progressive_mcus |
246 | 12 | )? |
247 | | }; |
248 | | |
249 | | // process that width up until it's impossible. This is faster than allocation the |
250 | | // full components, which we skipped earlier. |
251 | 38.1k | if all_components_in_first_scan { |
252 | 30.8k | self.post_process( |
253 | 30.8k | pixels, |
254 | 30.8k | i, |
255 | 30.8k | mcu_height, |
256 | 30.8k | width, |
257 | 30.8k | padded_width, |
258 | 30.8k | &mut pixels_written, |
259 | 30.8k | &mut upsampler_scratch_space |
260 | 5 | )?; |
261 | 7.35k | } |
262 | | |
263 | 252 | match terminate { |
264 | 33.1k | McuContinuation::Ok => {} |
265 | 22 | McuContinuation::AnotherSos if all_components_in_first_scan => { |
266 | 22 | warn!("More than one SOS despite already having all components"); |
267 | 22 | return Ok(()); |
268 | | } |
269 | 230 | McuContinuation::AnotherSos => continue 'sos, |
270 | 4.25k | McuContinuation::InterScanMarker(marker) => { |
271 | | // Handle inter-scan markers (DHT/DQT/etc) uniformly here. |
272 | | // This keeps all marker handling in the outer loop. |
273 | 4.25k | if self.advance_to_next_sos(marker, &mut stream)? { |
274 | 4.14k | continue 'sos; |
275 | | } else { |
276 | | // Hit EOI |
277 | 14 | break; |
278 | | } |
279 | | } |
280 | | McuContinuation::Terminate => { |
281 | 567 | warn!("Got terminate signal, will not process further"); |
282 | 567 | pixels.get_mut(pixels_written..).map(|v| v.fill(128)); Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::decode_mcu_ycbcr_baseline::{closure#3}<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::decode_mcu_ycbcr_baseline::{closure#3}Line | Count | Source | 282 | 567 | pixels.get_mut(pixels_written..).map(|v| v.fill(128)); |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::decode_mcu_ycbcr_baseline::{closure#3} |
283 | 567 | return Ok(()); |
284 | | } |
285 | | } |
286 | | } |
287 | | |
288 | | // Breaks if we get here, looping only if we have restarted, i.e. found another SOS and |
289 | | // continued at `'sos'. |
290 | 320 | break; |
291 | | } |
292 | | |
293 | 320 | if !all_components_in_first_scan { |
294 | 42 | self.finish_baseline_decoding(&progressive_mcus, mcu_width, pixels)?; |
295 | 278 | } |
296 | | |
297 | | // it may happen that some images don't have the whole buffer |
298 | | // so we can't panic in case of that |
299 | | // assert_eq!(pixels_written, pixels.len()); |
300 | | |
301 | | // For UHD usecases that tie two images separating them with EOI and |
302 | | // SOI markers, it may happen that we do not reach this image end of image |
303 | | // So this ensures we reach it |
304 | | // Ensure we read EOI |
305 | 318 | if !stream.seen_eoi { |
306 | 118 | let marker = get_marker(&mut self.stream, &mut stream); |
307 | 118 | match marker { |
308 | 72 | Ok(_m) => { |
309 | 72 | trace!("Found marker {:?}", _m); |
310 | 72 | } |
311 | 46 | Err(_) => { |
312 | 46 | // ignore error |
313 | 46 | } |
314 | | } |
315 | 200 | } |
316 | | |
317 | | trace!("Finished decoding image"); |
318 | | |
319 | 318 | Ok(()) |
320 | 1.03k | } Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::decode_mcu_ycbcr_baseline <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::decode_mcu_ycbcr_baseline Line | Count | Source | 88 | 1.03k | pub(crate) fn decode_mcu_ycbcr_baseline( | 89 | 1.03k | &mut self, pixels: &mut [u8] | 90 | 1.03k | ) -> Result<(), DecodeErrors> { | 91 | 1.03k | setup_component_params(self)?; | 92 | | | 93 | | // check dc and AC tables | 94 | 1.03k | self.check_tables()?; | 95 | | | 96 | | let (mut mcu_width, mut mcu_height); | 97 | | | 98 | 1.03k | if self.is_interleaved { | 99 | | // set upsampling functions | 100 | 584 | self.set_upsampling()?; | 101 | | | 102 | 584 | mcu_width = self.mcu_x; | 103 | 584 | mcu_height = self.mcu_y; | 104 | 447 | } else { | 105 | 447 | // For non-interleaved images( (1*1) subsampling) | 106 | 447 | // number of MCU's are the widths (+7 to account for paddings) divided bu 8. | 107 | 447 | mcu_width = (self.info.width as usize + 7) / 8; | 108 | 447 | mcu_height = (self.info.height as usize + 7) / 8; | 109 | 447 | } | 110 | 1.03k | if self.is_interleaved | 111 | 584 | && self.input_colorspace.num_components() > 1 | 112 | 577 | && self.options.jpeg_get_out_colorspace().num_components() == 1 | 113 | 0 | && (self.info.sample_ratio == SampleRatios::V | 114 | 0 | || self.info.sample_ratio == SampleRatios::HV) | 115 | 0 | { | 116 | 0 | // For a specific set of images, e.g interleaved, | 117 | 0 | // when converting from YcbCr to grayscale, we need to | 118 | 0 | // take into account mcu height since the MCU decoding needs to take | 119 | 0 | // it into account for padding purposes and the post processor | 120 | 0 | // parses two rows per mcu width. | 121 | 0 | // | 122 | 0 | // set coeff to be 2 to ensure that we increment two rows | 123 | 0 | // for every mcu processed also | 124 | 0 | mcu_height *= self.v_max; | 125 | 0 | mcu_height /= self.h_max; | 126 | 0 | self.coeff = 2; | 127 | 1.03k | } | 128 | | | 129 | 1.03k | if self.input_colorspace == ColorSpace::Luma && self.is_interleaved { | 130 | 6 | warn!("Grayscale image with down-sampled component, resetting component details"); | 131 | 6 | | 132 | 6 | self.reset_params(); | 133 | 6 | | 134 | 6 | mcu_width = ((self.info.width + 7) / 8) as usize; | 135 | 6 | mcu_height = ((self.info.height + 7) / 8) as usize; | 136 | 1.02k | } | 137 | 1.03k | let width = usize::from(self.info.width); | 138 | | | 139 | 1.03k | let padded_width = calculate_padded_width(width, self.info.sample_ratio); | 140 | | | 141 | 1.03k | let mut stream = BitStream::new(); | 142 | 1.03k | let mut tmp = [0_i32; DCT_BLOCK]; | 143 | | | 144 | 1.03k | let comp_len = self.components.len(); | 145 | | | 146 | 3.44k | for (pos, comp) in self.components.iter_mut().enumerate() { | 147 | | // Allocate only needed components. | 148 | | // | 149 | | // For special colorspaces i.e YCCK and CMYK, just allocate all of the needed | 150 | | // components. | 151 | 3.44k | if min( | 152 | 3.44k | self.options.jpeg_get_out_colorspace().num_components() - 1, | 153 | 3.44k | pos | 154 | 3.44k | ) == pos | 155 | 2 | || comp_len == 4 | 156 | | // Special colorspace | 157 | 3.44k | { | 158 | 3.44k | // allocate enough space to hold a whole MCU width | 159 | 3.44k | // this means we should take into account sampling ratios | 160 | 3.44k | // `*8` is because each MCU spans 8 widths. | 161 | 3.44k | let len = comp.width_stride * comp.vertical_sample * 8; | 162 | 3.44k | | 163 | 3.44k | comp.needed = true; | 164 | 3.44k | comp.raw_coeff = vec![0; len]; | 165 | 3.44k | } else { | 166 | 0 | comp.needed = false; | 167 | 0 | } | 168 | | } | 169 | | | 170 | | // If all components are contained in the first scan of MCUs, then we can process into | 171 | | // (upsampled) pixels immediately after each MCU, for convenience we use each row of MCUS. | 172 | | // Otherwise, we must first wait until following SOS provide the remaining components. | 173 | 1.03k | let all_components_in_first_scan = usize::from(self.num_scans) == self.components.len(); | 174 | 1.03k | let mut progressive_mcus: [Vec<i16>; 4] = core::array::from_fn(|_| vec![]); | 175 | | | 176 | 1.03k | if !all_components_in_first_scan { | 177 | 780 | for (component, mcu) in self.components.iter().zip(&mut progressive_mcus) { | 178 | 780 | let len = mcu_width | 179 | 780 | * component.vertical_sample | 180 | 780 | * component.horizontal_sample | 181 | 780 | * mcu_height | 182 | 780 | * 64; | 183 | 780 | *mcu = vec![0; len]; | 184 | 780 | } | 185 | 836 | } | 186 | | | 187 | 1.03k | let mut pixels_written = 0; | 188 | | | 189 | 1.03k | let is_hv = usize::from(self.is_interleaved); | 190 | 1.03k | let upsampler_scratch_size = is_hv * self.components.iter().map(|x| x.width_stride).max().unwrap_or(0) * 8; | 191 | 1.03k | let mut upsampler_scratch_space = vec![0; upsampler_scratch_size]; | 192 | | | 193 | | 'sos: loop { | 194 | | trace!( | 195 | | "Baseline decoding of components: {:?}", | 196 | | &self.z_order[..usize::from(self.num_scans)] | 197 | | ); | 198 | | | 199 | | trace!("Decoding MCU width: {mcu_width}, height: {mcu_height}"); | 200 | | | 201 | 38.2k | for i in 0..mcu_height { | 202 | 38.2k | if stream.overread_by > 0 { | 203 | 42 | pixels.get_mut(pixels_written..).map(|v| v.fill(128)); | 204 | 42 | if self.options.strict_mode() { | 205 | 0 | return Err(DecodeErrors::FormatStatic("Premature end of buffer")); | 206 | 42 | }; | 207 | | | 208 | 42 | error!("Premature end of buffer"); | 209 | 42 | break; | 210 | 38.2k | } | 211 | | | 212 | | // decode a whole MCU width, | 213 | | // this takes into account interleaved components. | 214 | 38.2k | let terminate = if all_components_in_first_scan { | 215 | 30.8k | self.decode_mcu_width::<false>( | 216 | 30.8k | mcu_width, | 217 | 30.8k | i, | 218 | 30.8k | &mut tmp, | 219 | 30.8k | &mut stream, | 220 | 30.8k | &mut progressive_mcus | 221 | 9 | )? | 222 | | } else { | 223 | | /* NB: (cae). This code was added due to the issue at https://github.com/etemesi254/zune-image/issues/277 | 224 | | * | 225 | | * There is a particular set of images that interleave the start of scan (SOS) with the MCU, | 226 | | * E.g if it's a three component image, we have SOS->MCU ->SOS->MCU ->SOS->MCU | 227 | | * which presents a problem on decoding, we need to buffer the whole image before continuing since | 228 | | * we won't have a row containing all the component data which will be needed e.g for color conversion. | 229 | | * | 230 | | * The mechanisms is that we decode the whole image upfront, which goes against the normal | 231 | | * routine of decoding MCU width , so this requires more memory upfront than initial routines | 232 | | * but it is a single image out of the many corpuses that exist, so its fine. | 233 | | * (image in test-images/jpeg/sos_news.jpeg) | 234 | | | 235 | | * Code contributed by Aurelia Molzer (https://github.com/197g) | 236 | | | 237 | | * | 238 | | */ | 239 | | | 240 | 7.36k | self.decode_mcu_width::<true>( | 241 | 7.36k | mcu_width, | 242 | 7.36k | i, | 243 | 7.36k | &mut tmp, | 244 | 7.36k | &mut stream, | 245 | 7.36k | &mut progressive_mcus | 246 | 12 | )? | 247 | | }; | 248 | | | 249 | | // process that width up until it's impossible. This is faster than allocation the | 250 | | // full components, which we skipped earlier. | 251 | 38.1k | if all_components_in_first_scan { | 252 | 30.8k | self.post_process( | 253 | 30.8k | pixels, | 254 | 30.8k | i, | 255 | 30.8k | mcu_height, | 256 | 30.8k | width, | 257 | 30.8k | padded_width, | 258 | 30.8k | &mut pixels_written, | 259 | 30.8k | &mut upsampler_scratch_space | 260 | 5 | )?; | 261 | 7.35k | } | 262 | | | 263 | 252 | match terminate { | 264 | 33.1k | McuContinuation::Ok => {} | 265 | 22 | McuContinuation::AnotherSos if all_components_in_first_scan => { | 266 | 22 | warn!("More than one SOS despite already having all components"); | 267 | 22 | return Ok(()); | 268 | | } | 269 | 230 | McuContinuation::AnotherSos => continue 'sos, | 270 | 4.25k | McuContinuation::InterScanMarker(marker) => { | 271 | | // Handle inter-scan markers (DHT/DQT/etc) uniformly here. | 272 | | // This keeps all marker handling in the outer loop. | 273 | 4.25k | if self.advance_to_next_sos(marker, &mut stream)? { | 274 | 4.14k | continue 'sos; | 275 | | } else { | 276 | | // Hit EOI | 277 | 14 | break; | 278 | | } | 279 | | } | 280 | | McuContinuation::Terminate => { | 281 | 567 | warn!("Got terminate signal, will not process further"); | 282 | 567 | pixels.get_mut(pixels_written..).map(|v| v.fill(128)); | 283 | 567 | return Ok(()); | 284 | | } | 285 | | } | 286 | | } | 287 | | | 288 | | // Breaks if we get here, looping only if we have restarted, i.e. found another SOS and | 289 | | // continued at `'sos'. | 290 | 320 | break; | 291 | | } | 292 | | | 293 | 320 | if !all_components_in_first_scan { | 294 | 42 | self.finish_baseline_decoding(&progressive_mcus, mcu_width, pixels)?; | 295 | 278 | } | 296 | | | 297 | | // it may happen that some images don't have the whole buffer | 298 | | // so we can't panic in case of that | 299 | | // assert_eq!(pixels_written, pixels.len()); | 300 | | | 301 | | // For UHD usecases that tie two images separating them with EOI and | 302 | | // SOI markers, it may happen that we do not reach this image end of image | 303 | | // So this ensures we reach it | 304 | | // Ensure we read EOI | 305 | 318 | if !stream.seen_eoi { | 306 | 118 | let marker = get_marker(&mut self.stream, &mut stream); | 307 | 118 | match marker { | 308 | 72 | Ok(_m) => { | 309 | 72 | trace!("Found marker {:?}", _m); | 310 | 72 | } | 311 | 46 | Err(_) => { | 312 | 46 | // ignore error | 313 | 46 | } | 314 | | } | 315 | 200 | } | 316 | | | 317 | | trace!("Finished decoding image"); | 318 | | | 319 | 318 | Ok(()) | 320 | 1.03k | } |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::decode_mcu_ycbcr_baseline |
321 | | |
322 | | /// Process all MCUs when baseline decoding has been processing them component-after-component. |
323 | | /// For simplicity this assembles the dequantized blocks in the order that the post processing |
324 | | /// of an interleaved baseline decoding would use. |
325 | | #[allow(clippy::too_many_lines)] |
326 | | #[allow(clippy::cast_sign_loss)] |
327 | 42 | pub(crate) fn finish_baseline_decoding( |
328 | 42 | &mut self, block: &[Vec<i16>; MAX_COMPONENTS], _mcu_width: usize, pixels: &mut [u8] |
329 | 42 | ) -> Result<(), DecodeErrors> { |
330 | 42 | let mcu_height = self.mcu_y; |
331 | | |
332 | | // Size of our output image(width*height) |
333 | 42 | let is_hv = usize::from(self.is_interleaved); |
334 | 42 | let upsampler_scratch_size = is_hv * self.components[0].width_stride; |
335 | 42 | let width = usize::from(self.info.width); |
336 | 42 | let padded_width = calculate_padded_width(width, self.info.sample_ratio); |
337 | | |
338 | 42 | let mut upsampler_scratch_space = vec![0; upsampler_scratch_size]; |
339 | | |
340 | 168 | for (pos, comp) in self.components.iter_mut().enumerate() { |
341 | | // Mark only needed components for computing output colors. |
342 | 168 | if min( |
343 | 168 | self.options.jpeg_get_out_colorspace().num_components() - 1, |
344 | 168 | pos |
345 | 168 | ) == pos |
346 | 1 | || self.input_colorspace == ColorSpace::YCCK |
347 | 1 | || self.input_colorspace == ColorSpace::CMYK |
348 | 167 | { |
349 | 167 | comp.needed = true; |
350 | 167 | } else { |
351 | 1 | comp.needed = false; |
352 | 1 | } |
353 | | } |
354 | | |
355 | 42 | let mut pixels_written = 0; |
356 | | |
357 | | // dequantize and idct have been performed, only color convert. |
358 | 11.9k | for i in 0..mcu_height { |
359 | | // All the data is already in the right order, we just need to be able to pass it to |
360 | | // the post_process & upsample method. That expects all the data to be stored as one |
361 | | // row of MCUs in each component's `raw_coeff`. |
362 | 47.8k | 'component: for (position, component) in &mut self.components.iter_mut().enumerate() { |
363 | 47.8k | if !component.needed { |
364 | 2 | continue 'component; |
365 | 47.8k | } |
366 | | |
367 | | // step is the number of pixels this iteration wil be handling |
368 | | // Given by the number of mcu's height and the length of the component block |
369 | | // Since the component block contains the whole channel as raw pixels |
370 | | // we this evenly divides the pixels into MCU blocks |
371 | | // |
372 | | // For interleaved images, this gives us the exact pixels comprising a whole MCU |
373 | | // block |
374 | 47.8k | let step = block[position].len() / mcu_height; |
375 | | |
376 | | // where we will be reading our pixels from. |
377 | 47.8k | let slice = &block[position][i * step..][..step]; |
378 | 47.8k | let temp_channel = &mut component.raw_coeff; |
379 | 47.8k | temp_channel[..step].copy_from_slice(slice); |
380 | | } |
381 | | |
382 | | // process that whole stripe of MCUs |
383 | 11.9k | self.post_process( |
384 | 11.9k | pixels, |
385 | 11.9k | i, |
386 | 11.9k | mcu_height, |
387 | 11.9k | width, |
388 | 11.9k | padded_width, |
389 | 11.9k | &mut pixels_written, |
390 | 11.9k | &mut upsampler_scratch_space |
391 | 2 | )?; |
392 | | } |
393 | | |
394 | 40 | return Ok(()); |
395 | 42 | } Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::finish_baseline_decoding <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::finish_baseline_decoding Line | Count | Source | 327 | 42 | pub(crate) fn finish_baseline_decoding( | 328 | 42 | &mut self, block: &[Vec<i16>; MAX_COMPONENTS], _mcu_width: usize, pixels: &mut [u8] | 329 | 42 | ) -> Result<(), DecodeErrors> { | 330 | 42 | let mcu_height = self.mcu_y; | 331 | | | 332 | | // Size of our output image(width*height) | 333 | 42 | let is_hv = usize::from(self.is_interleaved); | 334 | 42 | let upsampler_scratch_size = is_hv * self.components[0].width_stride; | 335 | 42 | let width = usize::from(self.info.width); | 336 | 42 | let padded_width = calculate_padded_width(width, self.info.sample_ratio); | 337 | | | 338 | 42 | let mut upsampler_scratch_space = vec![0; upsampler_scratch_size]; | 339 | | | 340 | 168 | for (pos, comp) in self.components.iter_mut().enumerate() { | 341 | | // Mark only needed components for computing output colors. | 342 | 168 | if min( | 343 | 168 | self.options.jpeg_get_out_colorspace().num_components() - 1, | 344 | 168 | pos | 345 | 168 | ) == pos | 346 | 1 | || self.input_colorspace == ColorSpace::YCCK | 347 | 1 | || self.input_colorspace == ColorSpace::CMYK | 348 | 167 | { | 349 | 167 | comp.needed = true; | 350 | 167 | } else { | 351 | 1 | comp.needed = false; | 352 | 1 | } | 353 | | } | 354 | | | 355 | 42 | let mut pixels_written = 0; | 356 | | | 357 | | // dequantize and idct have been performed, only color convert. | 358 | 11.9k | for i in 0..mcu_height { | 359 | | // All the data is already in the right order, we just need to be able to pass it to | 360 | | // the post_process & upsample method. That expects all the data to be stored as one | 361 | | // row of MCUs in each component's `raw_coeff`. | 362 | 47.8k | 'component: for (position, component) in &mut self.components.iter_mut().enumerate() { | 363 | 47.8k | if !component.needed { | 364 | 2 | continue 'component; | 365 | 47.8k | } | 366 | | | 367 | | // step is the number of pixels this iteration wil be handling | 368 | | // Given by the number of mcu's height and the length of the component block | 369 | | // Since the component block contains the whole channel as raw pixels | 370 | | // we this evenly divides the pixels into MCU blocks | 371 | | // | 372 | | // For interleaved images, this gives us the exact pixels comprising a whole MCU | 373 | | // block | 374 | 47.8k | let step = block[position].len() / mcu_height; | 375 | | | 376 | | // where we will be reading our pixels from. | 377 | 47.8k | let slice = &block[position][i * step..][..step]; | 378 | 47.8k | let temp_channel = &mut component.raw_coeff; | 379 | 47.8k | temp_channel[..step].copy_from_slice(slice); | 380 | | } | 381 | | | 382 | | // process that whole stripe of MCUs | 383 | 11.9k | self.post_process( | 384 | 11.9k | pixels, | 385 | 11.9k | i, | 386 | 11.9k | mcu_height, | 387 | 11.9k | width, | 388 | 11.9k | padded_width, | 389 | 11.9k | &mut pixels_written, | 390 | 11.9k | &mut upsampler_scratch_space | 391 | 2 | )?; | 392 | | } | 393 | | | 394 | 40 | return Ok(()); | 395 | 42 | } |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::finish_baseline_decoding |
396 | | |
397 | 38.2k | fn decode_mcu_width<const PROGRESSIVE: bool>( |
398 | 38.2k | &mut self, mcu_width: usize, mcu_height: usize, tmp: &mut [i32; 64], |
399 | 38.2k | stream: &mut BitStream, progressive: &mut [Vec<i16>; 4] |
400 | 38.2k | ) -> Result<McuContinuation, DecodeErrors> { |
401 | 38.2k | let is_one_by_one = !self.scan_subsampled; |
402 | | |
403 | | // The definition of MCU depends on the sampling factor of involved scans. When components |
404 | | // have different factors then each Minimal-Coding-Unit is the least common multiple such |
405 | | // that we have an integer number of blocks from each component. But the decoding of these |
406 | | // components differs from it otherwise, we need an inner loop with a dynamic amount of |
407 | | // coefficients per component, whereas otherwise we have exactly one block of coefficients |
408 | | // encoded for each component in the bitstream order. |
409 | | // |
410 | | // We statically specialize on this to improve code generation of the common case a little |
411 | | // bit. We could also special case common sub-sampling cases but be mindful of code bloat. |
412 | 38.2k | if is_one_by_one { |
413 | 30.1k | self.inner_decode_mcu_width::<PROGRESSIVE, false>( |
414 | 30.1k | mcu_width, |
415 | 30.1k | mcu_height, |
416 | 30.1k | tmp, |
417 | 30.1k | stream, |
418 | 30.1k | progressive |
419 | | ) |
420 | | } else { |
421 | 8.03k | self.inner_decode_mcu_width::<PROGRESSIVE, true>( |
422 | 8.03k | mcu_width, |
423 | 8.03k | mcu_height, |
424 | 8.03k | tmp, |
425 | 8.03k | stream, |
426 | 8.03k | progressive |
427 | | ) |
428 | | } |
429 | 38.2k | } Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::decode_mcu_width::<false> Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::decode_mcu_width::<true> <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::decode_mcu_width::<false> Line | Count | Source | 397 | 30.8k | fn decode_mcu_width<const PROGRESSIVE: bool>( | 398 | 30.8k | &mut self, mcu_width: usize, mcu_height: usize, tmp: &mut [i32; 64], | 399 | 30.8k | stream: &mut BitStream, progressive: &mut [Vec<i16>; 4] | 400 | 30.8k | ) -> Result<McuContinuation, DecodeErrors> { | 401 | 30.8k | let is_one_by_one = !self.scan_subsampled; | 402 | | | 403 | | // The definition of MCU depends on the sampling factor of involved scans. When components | 404 | | // have different factors then each Minimal-Coding-Unit is the least common multiple such | 405 | | // that we have an integer number of blocks from each component. But the decoding of these | 406 | | // components differs from it otherwise, we need an inner loop with a dynamic amount of | 407 | | // coefficients per component, whereas otherwise we have exactly one block of coefficients | 408 | | // encoded for each component in the bitstream order. | 409 | | // | 410 | | // We statically specialize on this to improve code generation of the common case a little | 411 | | // bit. We could also special case common sub-sampling cases but be mindful of code bloat. | 412 | 30.8k | if is_one_by_one { | 413 | 24.5k | self.inner_decode_mcu_width::<PROGRESSIVE, false>( | 414 | 24.5k | mcu_width, | 415 | 24.5k | mcu_height, | 416 | 24.5k | tmp, | 417 | 24.5k | stream, | 418 | 24.5k | progressive | 419 | | ) | 420 | | } else { | 421 | 6.26k | self.inner_decode_mcu_width::<PROGRESSIVE, true>( | 422 | 6.26k | mcu_width, | 423 | 6.26k | mcu_height, | 424 | 6.26k | tmp, | 425 | 6.26k | stream, | 426 | 6.26k | progressive | 427 | | ) | 428 | | } | 429 | 30.8k | } |
<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::decode_mcu_width::<true> Line | Count | Source | 397 | 7.36k | fn decode_mcu_width<const PROGRESSIVE: bool>( | 398 | 7.36k | &mut self, mcu_width: usize, mcu_height: usize, tmp: &mut [i32; 64], | 399 | 7.36k | stream: &mut BitStream, progressive: &mut [Vec<i16>; 4] | 400 | 7.36k | ) -> Result<McuContinuation, DecodeErrors> { | 401 | 7.36k | let is_one_by_one = !self.scan_subsampled; | 402 | | | 403 | | // The definition of MCU depends on the sampling factor of involved scans. When components | 404 | | // have different factors then each Minimal-Coding-Unit is the least common multiple such | 405 | | // that we have an integer number of blocks from each component. But the decoding of these | 406 | | // components differs from it otherwise, we need an inner loop with a dynamic amount of | 407 | | // coefficients per component, whereas otherwise we have exactly one block of coefficients | 408 | | // encoded for each component in the bitstream order. | 409 | | // | 410 | | // We statically specialize on this to improve code generation of the common case a little | 411 | | // bit. We could also special case common sub-sampling cases but be mindful of code bloat. | 412 | 7.36k | if is_one_by_one { | 413 | 5.59k | self.inner_decode_mcu_width::<PROGRESSIVE, false>( | 414 | 5.59k | mcu_width, | 415 | 5.59k | mcu_height, | 416 | 5.59k | tmp, | 417 | 5.59k | stream, | 418 | 5.59k | progressive | 419 | | ) | 420 | | } else { | 421 | 1.77k | self.inner_decode_mcu_width::<PROGRESSIVE, true>( | 422 | 1.77k | mcu_width, | 423 | 1.77k | mcu_height, | 424 | 1.77k | tmp, | 425 | 1.77k | stream, | 426 | 1.77k | progressive | 427 | | ) | 428 | | } | 429 | 7.36k | } |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::decode_mcu_width::<_> |
430 | | |
431 | | // Inline-never ensures we do get this function optimize on its own, into two different |
432 | | // versions, without the optimizer tripping up over the complexity that comes with the |
433 | | // constant folding. And constant folding is quite important for performance here as |
434 | | // when `not SAMPLED` then the inner loop has exactly one iteration per component in |
435 | | // the scan. The difference was ~1% or a bit more. |
436 | 38.2k | fn inner_decode_mcu_width<const PROGRESSIVE: bool, const SAMPLED: bool>( |
437 | 38.2k | &mut self, mcu_width: usize, mcu_height: usize, tmp: &mut [i32; 64], |
438 | 38.2k | stream: &mut BitStream, progressive: &mut [Vec<i16>; 4] |
439 | 38.2k | ) -> Result<McuContinuation, DecodeErrors> { |
440 | 38.2k | let z_order = self.z_order; |
441 | 38.2k | let z_scans = &z_order[..usize::from(self.num_scans)]; |
442 | | |
443 | | // How much of the head of `tmp` was written by the last MCU decoding? We only check for |
444 | | // two different cases and not all possible outcomes as this is only used to optimize the |
445 | | // bytes written in `fill`. Since the clobber happens in UNZIGZAG order we'd be straddling |
446 | | // most cache lines anyways even if we did a partial write with the exact length of the |
447 | | // coefficient data which was written into `tmp`. |
448 | 38.2k | let mut clobber_more_than_4x4 = true; |
449 | | |
450 | | // For non-interleaved scans (PROGRESSIVE=true), each scan contains a single component |
451 | | // and we iterate over that component's actual data unit count, not the interleaved MCU |
452 | | // width multiplied by sampling factor. |
453 | 38.2k | let mut scan_du_width = if PROGRESSIVE { |
454 | 7.36k | let k = z_scans[0]; |
455 | 7.36k | let comp = &self.components[k]; |
456 | | // Calculate actual data units for this component: ceil(width / (8 * subsampling_ratio)) |
457 | 7.36k | (self.info.width as usize * comp.horizontal_sample + self.h_max * 8 - 1) |
458 | 7.36k | / (self.h_max * 8) |
459 | | } else { |
460 | 30.8k | mcu_width |
461 | | }; |
462 | | // In malformed scans that list multiple components, clamp to the smallest row capacity |
463 | | // to avoid writing past the row buffer. |
464 | 38.2k | if PROGRESSIVE && z_scans.len() > 1 { |
465 | 1.41k | let min_du = z_scans |
466 | 1.41k | .iter() |
467 | 5.64k | .map(|&k| self.components[k].width_stride / 8) Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::inner_decode_mcu_width::<true, true>::{closure#0}Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::inner_decode_mcu_width::<true, false>::{closure#0}<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::inner_decode_mcu_width::<true, true>::{closure#0}Line | Count | Source | 467 | 5.64k | .map(|&k| self.components[k].width_stride / 8) |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::inner_decode_mcu_width::<true, false>::{closure#0}Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::inner_decode_mcu_width::<_, _>::{closure#0} |
468 | 1.41k | .min() |
469 | 1.41k | .unwrap_or(0); |
470 | 1.41k | scan_du_width = scan_du_width.min(min_du); |
471 | 36.8k | } |
472 | | |
473 | 1.10M | for j in 0..scan_du_width { |
474 | | // iterate over components |
475 | 3.75M | for &k in z_scans { |
476 | | // we made this loop body massive due to several different paths that depend on |
477 | | // static conditions. Note we (potentially) call into other functions so the |
478 | | // compiler will not unroll anything here anyways. The gains from separating |
479 | | // differently optimized loop bodies are much greater than a single additional jump |
480 | | // here. |
481 | 2.64M | let component = &mut self.components[k]; |
482 | | |
483 | 2.64M | let dc_table = self.dc_huffman_tables[component.dc_huff_table % MAX_COMPONENTS] |
484 | 2.64M | .as_ref() |
485 | 2.64M | .ok_or(DecodeErrors::FormatStatic("DC table not found"))?; |
486 | | |
487 | 2.64M | let ac_table = self.ac_huffman_tables[component.ac_huff_table % MAX_COMPONENTS] |
488 | 2.64M | .as_ref() |
489 | 2.64M | .ok_or(DecodeErrors::FormatStatic("AC table not found"))?; |
490 | | |
491 | 2.64M | let qt_table = &component.quantization_table; |
492 | 2.64M | let channel = if PROGRESSIVE { |
493 | 1.41M | let offset = |
494 | 1.41M | mcu_height * component.width_stride * 8 * component.vertical_sample; |
495 | | // Small stopgap for https://github.com/etemesi254/zune-image/issues/362 |
496 | 1.41M | if offset >= progressive[k].len(){ |
497 | 1 | return Err(DecodeErrors::FormatStatic("Would panic on slice iteration")) |
498 | 1.41M | } |
499 | 1.41M | &mut progressive[k][offset..] |
500 | | } else { |
501 | 1.22M | &mut component.raw_coeff |
502 | | }; |
503 | | |
504 | 2.64M | let component_samples_needed = component.needed; |
505 | | |
506 | | // If image is interleaved iterate over scan components, |
507 | | // otherwise if it-s non-interleaved, these routines iterate in |
508 | | // trivial scanline order(Y,Cb,Cr) |
509 | | // |
510 | | // Turn the bounds into a compile time constant for a common special case. This |
511 | | // allows the compiler to unroll the loop and then do a bunch of interleaving. |
512 | | // |
513 | | // For PROGRESSIVE (non-interleaved), we iterate data units directly so |
514 | | // h_samp/v_samp loops run exactly once. |
515 | 2.64M | let v_step = |
516 | 2.64M | if SAMPLED && !PROGRESSIVE { 0..component.vertical_sample } else { 0..1 }; |
517 | | |
518 | 5.73M | for v_samp in v_step { |
519 | 3.08M | let h_step = |
520 | 3.08M | if SAMPLED && !PROGRESSIVE { 0..component.horizontal_sample } else { 0..1 }; |
521 | | |
522 | 6.18M | for h_samp in h_step { |
523 | 3.10M | let result = if component_samples_needed { |
524 | | // Fill the array with zeroes, decode_mcu_block expects |
525 | | // a zero based array. Clobber is in zig-zag order though. |
526 | | // Writing consecutive entries is basically free in terms |
527 | | // of memory throughput so we opt for a larger power of |
528 | | // two which lets the compiler turn this into a repeated |
529 | | // write of a zeroed vector register, which does not have |
530 | | // any branches, instead of a more difficult pattern where |
531 | | // we attempt to overwrite exactly one coefficient. |
532 | 3.10M | let clobber_len = if !clobber_more_than_4x4 { 32 } else { 64 }; |
533 | | |
534 | 3.10M | tmp[..clobber_len].fill(0); |
535 | | |
536 | 3.10M | stream.decode_mcu_block( |
537 | 3.10M | &mut self.stream, |
538 | 3.10M | dc_table, |
539 | 3.10M | ac_table, |
540 | 3.10M | qt_table, |
541 | 3.10M | tmp, |
542 | 3.10M | &mut component.dc_pred |
543 | | ) |
544 | | } else { |
545 | | // We do not touch tmp so there is no need to reset it. |
546 | 0 | stream.discard_mcu_block(&mut self.stream, dc_table, ac_table) |
547 | | }; |
548 | | |
549 | | // If an error occurs we can either propagate it |
550 | | // as an error or print it and call terminate. |
551 | | // |
552 | | // This allows even corrupt images to render something, |
553 | | // even if its bad, matching browsers. |
554 | | // |
555 | | // See example in https://github.com/etemesi254/zune-image/issues/293 |
556 | 3.10M | let len = if let Ok(len) = result { |
557 | 3.10M | len |
558 | | } else { |
559 | | // result.is_err() |
560 | 560 | return if self.options.strict_mode() { |
561 | 0 | Err(result.err().unwrap()) |
562 | | } else { |
563 | 560 | error!("{}", result.err().unwrap()); |
564 | 560 | Ok(McuContinuation::Terminate) |
565 | | }; |
566 | | }; |
567 | | |
568 | 3.10M | if component_samples_needed { |
569 | | // tmp was only written partially, note that len is in ZigZag order. |
570 | 3.10M | clobber_more_than_4x4 = len > 10; |
571 | | |
572 | 3.10M | let idct_position = if PROGRESSIVE { |
573 | | // For non-interleaved, j indexes data units directly |
574 | 1.41M | j * 8 |
575 | | } else { |
576 | | // derived from stb and rewritten for my tastes |
577 | 1.68M | let c2 = v_samp * 8; |
578 | 1.68M | let c3 = ((j * component.horizontal_sample) + h_samp) * 8; |
579 | | |
580 | 1.68M | component.width_stride * c2 + c3 |
581 | | }; |
582 | | |
583 | 3.10M | let idct_pos = channel.get_mut(idct_position..).unwrap(); |
584 | | |
585 | 3.10M | if len <= 1 { |
586 | 713k | (self.idct_1x1_func)(tmp, idct_pos, component.width_stride); |
587 | 2.38M | } else if len <= 10 { |
588 | 21.0k | (self.idct_4x4_func)(tmp, idct_pos, component.width_stride); |
589 | 2.36M | } else { |
590 | 2.36M | // call idct. |
591 | 2.36M | (self.idct_func)(tmp, idct_pos, component.width_stride); |
592 | 2.36M | } |
593 | 0 | } |
594 | | } |
595 | | } |
596 | | } |
597 | | |
598 | 1.10M | self.todo = self.todo.wrapping_sub(1); |
599 | | |
600 | 1.10M | if self.todo == 0 { |
601 | 2.55k | self.handle_rst_main(stream)?; |
602 | 2.55k | continue; |
603 | 1.10M | } |
604 | | |
605 | 1.10M | if stream.marker.is_some() && stream.bits_left == 0 { |
606 | 37 | break; |
607 | 1.10M | } |
608 | | } |
609 | | |
610 | 37.6k | self.check_stream_marker_after_mcu_width(stream) |
611 | 38.2k | } Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::inner_decode_mcu_width::<false, false> Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::inner_decode_mcu_width::<false, true> Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::inner_decode_mcu_width::<true, true> Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::inner_decode_mcu_width::<true, false> <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::inner_decode_mcu_width::<false, false> Line | Count | Source | 436 | 24.5k | fn inner_decode_mcu_width<const PROGRESSIVE: bool, const SAMPLED: bool>( | 437 | 24.5k | &mut self, mcu_width: usize, mcu_height: usize, tmp: &mut [i32; 64], | 438 | 24.5k | stream: &mut BitStream, progressive: &mut [Vec<i16>; 4] | 439 | 24.5k | ) -> Result<McuContinuation, DecodeErrors> { | 440 | 24.5k | let z_order = self.z_order; | 441 | 24.5k | let z_scans = &z_order[..usize::from(self.num_scans)]; | 442 | | | 443 | | // How much of the head of `tmp` was written by the last MCU decoding? We only check for | 444 | | // two different cases and not all possible outcomes as this is only used to optimize the | 445 | | // bytes written in `fill`. Since the clobber happens in UNZIGZAG order we'd be straddling | 446 | | // most cache lines anyways even if we did a partial write with the exact length of the | 447 | | // coefficient data which was written into `tmp`. | 448 | 24.5k | let mut clobber_more_than_4x4 = true; | 449 | | | 450 | | // For non-interleaved scans (PROGRESSIVE=true), each scan contains a single component | 451 | | // and we iterate over that component's actual data unit count, not the interleaved MCU | 452 | | // width multiplied by sampling factor. | 453 | 24.5k | let mut scan_du_width = if PROGRESSIVE { | 454 | 0 | let k = z_scans[0]; | 455 | 0 | let comp = &self.components[k]; | 456 | | // Calculate actual data units for this component: ceil(width / (8 * subsampling_ratio)) | 457 | 0 | (self.info.width as usize * comp.horizontal_sample + self.h_max * 8 - 1) | 458 | 0 | / (self.h_max * 8) | 459 | | } else { | 460 | 24.5k | mcu_width | 461 | | }; | 462 | | // In malformed scans that list multiple components, clamp to the smallest row capacity | 463 | | // to avoid writing past the row buffer. | 464 | 24.5k | if PROGRESSIVE && z_scans.len() > 1 { | 465 | 0 | let min_du = z_scans | 466 | 0 | .iter() | 467 | 0 | .map(|&k| self.components[k].width_stride / 8) | 468 | 0 | .min() | 469 | 0 | .unwrap_or(0); | 470 | 0 | scan_du_width = scan_du_width.min(min_du); | 471 | 24.5k | } | 472 | | | 473 | 299k | for j in 0..scan_du_width { | 474 | | // iterate over components | 475 | 643k | for &k in z_scans { | 476 | | // we made this loop body massive due to several different paths that depend on | 477 | | // static conditions. Note we (potentially) call into other functions so the | 478 | | // compiler will not unroll anything here anyways. The gains from separating | 479 | | // differently optimized loop bodies are much greater than a single additional jump | 480 | | // here. | 481 | 344k | let component = &mut self.components[k]; | 482 | | | 483 | 344k | let dc_table = self.dc_huffman_tables[component.dc_huff_table % MAX_COMPONENTS] | 484 | 344k | .as_ref() | 485 | 344k | .ok_or(DecodeErrors::FormatStatic("DC table not found"))?; | 486 | | | 487 | 344k | let ac_table = self.ac_huffman_tables[component.ac_huff_table % MAX_COMPONENTS] | 488 | 344k | .as_ref() | 489 | 344k | .ok_or(DecodeErrors::FormatStatic("AC table not found"))?; | 490 | | | 491 | 344k | let qt_table = &component.quantization_table; | 492 | 344k | let channel = if PROGRESSIVE { | 493 | 0 | let offset = | 494 | 0 | mcu_height * component.width_stride * 8 * component.vertical_sample; | 495 | | // Small stopgap for https://github.com/etemesi254/zune-image/issues/362 | 496 | 0 | if offset >= progressive[k].len(){ | 497 | 0 | return Err(DecodeErrors::FormatStatic("Would panic on slice iteration")) | 498 | 0 | } | 499 | 0 | &mut progressive[k][offset..] | 500 | | } else { | 501 | 344k | &mut component.raw_coeff | 502 | | }; | 503 | | | 504 | 344k | let component_samples_needed = component.needed; | 505 | | | 506 | | // If image is interleaved iterate over scan components, | 507 | | // otherwise if it-s non-interleaved, these routines iterate in | 508 | | // trivial scanline order(Y,Cb,Cr) | 509 | | // | 510 | | // Turn the bounds into a compile time constant for a common special case. This | 511 | | // allows the compiler to unroll the loop and then do a bunch of interleaving. | 512 | | // | 513 | | // For PROGRESSIVE (non-interleaved), we iterate data units directly so | 514 | | // h_samp/v_samp loops run exactly once. | 515 | 344k | let v_step = | 516 | 344k | if SAMPLED && !PROGRESSIVE { 0..component.vertical_sample } else { 0..1 }; | 517 | | | 518 | 688k | for v_samp in v_step { | 519 | 344k | let h_step = | 520 | 344k | if SAMPLED && !PROGRESSIVE { 0..component.horizontal_sample } else { 0..1 }; | 521 | | | 522 | 688k | for h_samp in h_step { | 523 | 344k | let result = if component_samples_needed { | 524 | | // Fill the array with zeroes, decode_mcu_block expects | 525 | | // a zero based array. Clobber is in zig-zag order though. | 526 | | // Writing consecutive entries is basically free in terms | 527 | | // of memory throughput so we opt for a larger power of | 528 | | // two which lets the compiler turn this into a repeated | 529 | | // write of a zeroed vector register, which does not have | 530 | | // any branches, instead of a more difficult pattern where | 531 | | // we attempt to overwrite exactly one coefficient. | 532 | 344k | let clobber_len = if !clobber_more_than_4x4 { 32 } else { 64 }; | 533 | | | 534 | 344k | tmp[..clobber_len].fill(0); | 535 | | | 536 | 344k | stream.decode_mcu_block( | 537 | 344k | &mut self.stream, | 538 | 344k | dc_table, | 539 | 344k | ac_table, | 540 | 344k | qt_table, | 541 | 344k | tmp, | 542 | 344k | &mut component.dc_pred | 543 | | ) | 544 | | } else { | 545 | | // We do not touch tmp so there is no need to reset it. | 546 | 0 | stream.discard_mcu_block(&mut self.stream, dc_table, ac_table) | 547 | | }; | 548 | | | 549 | | // If an error occurs we can either propagate it | 550 | | // as an error or print it and call terminate. | 551 | | // | 552 | | // This allows even corrupt images to render something, | 553 | | // even if its bad, matching browsers. | 554 | | // | 555 | | // See example in https://github.com/etemesi254/zune-image/issues/293 | 556 | 344k | let len = if let Ok(len) = result { | 557 | 344k | len | 558 | | } else { | 559 | | // result.is_err() | 560 | 247 | return if self.options.strict_mode() { | 561 | 0 | Err(result.err().unwrap()) | 562 | | } else { | 563 | 247 | error!("{}", result.err().unwrap()); | 564 | 247 | Ok(McuContinuation::Terminate) | 565 | | }; | 566 | | }; | 567 | | | 568 | 344k | if component_samples_needed { | 569 | | // tmp was only written partially, note that len is in ZigZag order. | 570 | 344k | clobber_more_than_4x4 = len > 10; | 571 | | | 572 | 344k | let idct_position = if PROGRESSIVE { | 573 | | // For non-interleaved, j indexes data units directly | 574 | 0 | j * 8 | 575 | | } else { | 576 | | // derived from stb and rewritten for my tastes | 577 | 344k | let c2 = v_samp * 8; | 578 | 344k | let c3 = ((j * component.horizontal_sample) + h_samp) * 8; | 579 | | | 580 | 344k | component.width_stride * c2 + c3 | 581 | | }; | 582 | | | 583 | 344k | let idct_pos = channel.get_mut(idct_position..).unwrap(); | 584 | | | 585 | 344k | if len <= 1 { | 586 | 193k | (self.idct_1x1_func)(tmp, idct_pos, component.width_stride); | 587 | 193k | } else if len <= 10 { | 588 | 11.7k | (self.idct_4x4_func)(tmp, idct_pos, component.width_stride); | 589 | 138k | } else { | 590 | 138k | // call idct. | 591 | 138k | (self.idct_func)(tmp, idct_pos, component.width_stride); | 592 | 138k | } | 593 | 0 | } | 594 | | } | 595 | | } | 596 | | } | 597 | | | 598 | 298k | self.todo = self.todo.wrapping_sub(1); | 599 | | | 600 | 298k | if self.todo == 0 { | 601 | 1.17k | self.handle_rst_main(stream)?; | 602 | 1.17k | continue; | 603 | 297k | } | 604 | | | 605 | 297k | if stream.marker.is_some() && stream.bits_left == 0 { | 606 | 35 | break; | 607 | 297k | } | 608 | | } | 609 | | | 610 | 24.3k | self.check_stream_marker_after_mcu_width(stream) | 611 | 24.5k | } |
<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::inner_decode_mcu_width::<false, true> Line | Count | Source | 436 | 6.26k | fn inner_decode_mcu_width<const PROGRESSIVE: bool, const SAMPLED: bool>( | 437 | 6.26k | &mut self, mcu_width: usize, mcu_height: usize, tmp: &mut [i32; 64], | 438 | 6.26k | stream: &mut BitStream, progressive: &mut [Vec<i16>; 4] | 439 | 6.26k | ) -> Result<McuContinuation, DecodeErrors> { | 440 | 6.26k | let z_order = self.z_order; | 441 | 6.26k | let z_scans = &z_order[..usize::from(self.num_scans)]; | 442 | | | 443 | | // How much of the head of `tmp` was written by the last MCU decoding? We only check for | 444 | | // two different cases and not all possible outcomes as this is only used to optimize the | 445 | | // bytes written in `fill`. Since the clobber happens in UNZIGZAG order we'd be straddling | 446 | | // most cache lines anyways even if we did a partial write with the exact length of the | 447 | | // coefficient data which was written into `tmp`. | 448 | 6.26k | let mut clobber_more_than_4x4 = true; | 449 | | | 450 | | // For non-interleaved scans (PROGRESSIVE=true), each scan contains a single component | 451 | | // and we iterate over that component's actual data unit count, not the interleaved MCU | 452 | | // width multiplied by sampling factor. | 453 | 6.26k | let mut scan_du_width = if PROGRESSIVE { | 454 | 0 | let k = z_scans[0]; | 455 | 0 | let comp = &self.components[k]; | 456 | | // Calculate actual data units for this component: ceil(width / (8 * subsampling_ratio)) | 457 | 0 | (self.info.width as usize * comp.horizontal_sample + self.h_max * 8 - 1) | 458 | 0 | / (self.h_max * 8) | 459 | | } else { | 460 | 6.26k | mcu_width | 461 | | }; | 462 | | // In malformed scans that list multiple components, clamp to the smallest row capacity | 463 | | // to avoid writing past the row buffer. | 464 | 6.26k | if PROGRESSIVE && z_scans.len() > 1 { | 465 | 0 | let min_du = z_scans | 466 | 0 | .iter() | 467 | 0 | .map(|&k| self.components[k].width_stride / 8) | 468 | 0 | .min() | 469 | 0 | .unwrap_or(0); | 470 | 0 | scan_du_width = scan_du_width.min(min_du); | 471 | 6.26k | } | 472 | | | 473 | 224k | for j in 0..scan_du_width { | 474 | | // iterate over components | 475 | 1.10M | for &k in z_scans { | 476 | | // we made this loop body massive due to several different paths that depend on | 477 | | // static conditions. Note we (potentially) call into other functions so the | 478 | | // compiler will not unroll anything here anyways. The gains from separating | 479 | | // differently optimized loop bodies are much greater than a single additional jump | 480 | | // here. | 481 | 884k | let component = &mut self.components[k]; | 482 | | | 483 | 884k | let dc_table = self.dc_huffman_tables[component.dc_huff_table % MAX_COMPONENTS] | 484 | 884k | .as_ref() | 485 | 884k | .ok_or(DecodeErrors::FormatStatic("DC table not found"))?; | 486 | | | 487 | 884k | let ac_table = self.ac_huffman_tables[component.ac_huff_table % MAX_COMPONENTS] | 488 | 884k | .as_ref() | 489 | 884k | .ok_or(DecodeErrors::FormatStatic("AC table not found"))?; | 490 | | | 491 | 884k | let qt_table = &component.quantization_table; | 492 | 884k | let channel = if PROGRESSIVE { | 493 | 0 | let offset = | 494 | 0 | mcu_height * component.width_stride * 8 * component.vertical_sample; | 495 | | // Small stopgap for https://github.com/etemesi254/zune-image/issues/362 | 496 | 0 | if offset >= progressive[k].len(){ | 497 | 0 | return Err(DecodeErrors::FormatStatic("Would panic on slice iteration")) | 498 | 0 | } | 499 | 0 | &mut progressive[k][offset..] | 500 | | } else { | 501 | 884k | &mut component.raw_coeff | 502 | | }; | 503 | | | 504 | 884k | let component_samples_needed = component.needed; | 505 | | | 506 | | // If image is interleaved iterate over scan components, | 507 | | // otherwise if it-s non-interleaved, these routines iterate in | 508 | | // trivial scanline order(Y,Cb,Cr) | 509 | | // | 510 | | // Turn the bounds into a compile time constant for a common special case. This | 511 | | // allows the compiler to unroll the loop and then do a bunch of interleaving. | 512 | | // | 513 | | // For PROGRESSIVE (non-interleaved), we iterate data units directly so | 514 | | // h_samp/v_samp loops run exactly once. | 515 | 884k | let v_step = | 516 | 884k | if SAMPLED && !PROGRESSIVE { 0..component.vertical_sample } else { 0..1 }; | 517 | | | 518 | 2.20M | for v_samp in v_step { | 519 | 1.32M | let h_step = | 520 | 1.32M | if SAMPLED && !PROGRESSIVE { 0..component.horizontal_sample } else { 0..1 }; | 521 | | | 522 | 2.66M | for h_samp in h_step { | 523 | 1.33M | let result = if component_samples_needed { | 524 | | // Fill the array with zeroes, decode_mcu_block expects | 525 | | // a zero based array. Clobber is in zig-zag order though. | 526 | | // Writing consecutive entries is basically free in terms | 527 | | // of memory throughput so we opt for a larger power of | 528 | | // two which lets the compiler turn this into a repeated | 529 | | // write of a zeroed vector register, which does not have | 530 | | // any branches, instead of a more difficult pattern where | 531 | | // we attempt to overwrite exactly one coefficient. | 532 | 1.33M | let clobber_len = if !clobber_more_than_4x4 { 32 } else { 64 }; | 533 | | | 534 | 1.33M | tmp[..clobber_len].fill(0); | 535 | | | 536 | 1.33M | stream.decode_mcu_block( | 537 | 1.33M | &mut self.stream, | 538 | 1.33M | dc_table, | 539 | 1.33M | ac_table, | 540 | 1.33M | qt_table, | 541 | 1.33M | tmp, | 542 | 1.33M | &mut component.dc_pred | 543 | | ) | 544 | | } else { | 545 | | // We do not touch tmp so there is no need to reset it. | 546 | 0 | stream.discard_mcu_block(&mut self.stream, dc_table, ac_table) | 547 | | }; | 548 | | | 549 | | // If an error occurs we can either propagate it | 550 | | // as an error or print it and call terminate. | 551 | | // | 552 | | // This allows even corrupt images to render something, | 553 | | // even if its bad, matching browsers. | 554 | | // | 555 | | // See example in https://github.com/etemesi254/zune-image/issues/293 | 556 | 1.33M | let len = if let Ok(len) = result { | 557 | 1.33M | len | 558 | | } else { | 559 | | // result.is_err() | 560 | 213 | return if self.options.strict_mode() { | 561 | 0 | Err(result.err().unwrap()) | 562 | | } else { | 563 | 213 | error!("{}", result.err().unwrap()); | 564 | 213 | Ok(McuContinuation::Terminate) | 565 | | }; | 566 | | }; | 567 | | | 568 | 1.33M | if component_samples_needed { | 569 | | // tmp was only written partially, note that len is in ZigZag order. | 570 | 1.33M | clobber_more_than_4x4 = len > 10; | 571 | | | 572 | 1.33M | let idct_position = if PROGRESSIVE { | 573 | | // For non-interleaved, j indexes data units directly | 574 | 0 | j * 8 | 575 | | } else { | 576 | | // derived from stb and rewritten for my tastes | 577 | 1.33M | let c2 = v_samp * 8; | 578 | 1.33M | let c3 = ((j * component.horizontal_sample) + h_samp) * 8; | 579 | | | 580 | 1.33M | component.width_stride * c2 + c3 | 581 | | }; | 582 | | | 583 | 1.33M | let idct_pos = channel.get_mut(idct_position..).unwrap(); | 584 | | | 585 | 1.33M | if len <= 1 { | 586 | 238k | (self.idct_1x1_func)(tmp, idct_pos, component.width_stride); | 587 | 1.09M | } else if len <= 10 { | 588 | 2.63k | (self.idct_4x4_func)(tmp, idct_pos, component.width_stride); | 589 | 1.09M | } else { | 590 | 1.09M | // call idct. | 591 | 1.09M | (self.idct_func)(tmp, idct_pos, component.width_stride); | 592 | 1.09M | } | 593 | 0 | } | 594 | | } | 595 | | } | 596 | | } | 597 | | | 598 | 224k | self.todo = self.todo.wrapping_sub(1); | 599 | | | 600 | 224k | if self.todo == 0 { | 601 | 670 | self.handle_rst_main(stream)?; | 602 | 670 | continue; | 603 | 223k | } | 604 | | | 605 | 223k | if stream.marker.is_some() && stream.bits_left == 0 { | 606 | 2 | break; | 607 | 223k | } | 608 | | } | 609 | | | 610 | 6.04k | self.check_stream_marker_after_mcu_width(stream) | 611 | 6.26k | } |
<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::inner_decode_mcu_width::<true, true> Line | Count | Source | 436 | 1.77k | fn inner_decode_mcu_width<const PROGRESSIVE: bool, const SAMPLED: bool>( | 437 | 1.77k | &mut self, mcu_width: usize, mcu_height: usize, tmp: &mut [i32; 64], | 438 | 1.77k | stream: &mut BitStream, progressive: &mut [Vec<i16>; 4] | 439 | 1.77k | ) -> Result<McuContinuation, DecodeErrors> { | 440 | 1.77k | let z_order = self.z_order; | 441 | 1.77k | let z_scans = &z_order[..usize::from(self.num_scans)]; | 442 | | | 443 | | // How much of the head of `tmp` was written by the last MCU decoding? We only check for | 444 | | // two different cases and not all possible outcomes as this is only used to optimize the | 445 | | // bytes written in `fill`. Since the clobber happens in UNZIGZAG order we'd be straddling | 446 | | // most cache lines anyways even if we did a partial write with the exact length of the | 447 | | // coefficient data which was written into `tmp`. | 448 | 1.77k | let mut clobber_more_than_4x4 = true; | 449 | | | 450 | | // For non-interleaved scans (PROGRESSIVE=true), each scan contains a single component | 451 | | // and we iterate over that component's actual data unit count, not the interleaved MCU | 452 | | // width multiplied by sampling factor. | 453 | 1.77k | let mut scan_du_width = if PROGRESSIVE { | 454 | 1.77k | let k = z_scans[0]; | 455 | 1.77k | let comp = &self.components[k]; | 456 | | // Calculate actual data units for this component: ceil(width / (8 * subsampling_ratio)) | 457 | 1.77k | (self.info.width as usize * comp.horizontal_sample + self.h_max * 8 - 1) | 458 | 1.77k | / (self.h_max * 8) | 459 | | } else { | 460 | 0 | mcu_width | 461 | | }; | 462 | | // In malformed scans that list multiple components, clamp to the smallest row capacity | 463 | | // to avoid writing past the row buffer. | 464 | 1.77k | if PROGRESSIVE && z_scans.len() > 1 { | 465 | 1.41k | let min_du = z_scans | 466 | 1.41k | .iter() | 467 | 1.41k | .map(|&k| self.components[k].width_stride / 8) | 468 | 1.41k | .min() | 469 | 1.41k | .unwrap_or(0); | 470 | 1.41k | scan_du_width = scan_du_width.min(min_du); | 471 | 360 | } | 472 | | | 473 | 285k | for j in 0..scan_du_width { | 474 | | // iterate over components | 475 | 1.40M | for &k in z_scans { | 476 | | // we made this loop body massive due to several different paths that depend on | 477 | | // static conditions. Note we (potentially) call into other functions so the | 478 | | // compiler will not unroll anything here anyways. The gains from separating | 479 | | // differently optimized loop bodies are much greater than a single additional jump | 480 | | // here. | 481 | 1.12M | let component = &mut self.components[k]; | 482 | | | 483 | 1.12M | let dc_table = self.dc_huffman_tables[component.dc_huff_table % MAX_COMPONENTS] | 484 | 1.12M | .as_ref() | 485 | 1.12M | .ok_or(DecodeErrors::FormatStatic("DC table not found"))?; | 486 | | | 487 | 1.12M | let ac_table = self.ac_huffman_tables[component.ac_huff_table % MAX_COMPONENTS] | 488 | 1.12M | .as_ref() | 489 | 1.12M | .ok_or(DecodeErrors::FormatStatic("AC table not found"))?; | 490 | | | 491 | 1.12M | let qt_table = &component.quantization_table; | 492 | 1.12M | let channel = if PROGRESSIVE { | 493 | 1.12M | let offset = | 494 | 1.12M | mcu_height * component.width_stride * 8 * component.vertical_sample; | 495 | | // Small stopgap for https://github.com/etemesi254/zune-image/issues/362 | 496 | 1.12M | if offset >= progressive[k].len(){ | 497 | 1 | return Err(DecodeErrors::FormatStatic("Would panic on slice iteration")) | 498 | 1.12M | } | 499 | 1.12M | &mut progressive[k][offset..] | 500 | | } else { | 501 | 0 | &mut component.raw_coeff | 502 | | }; | 503 | | | 504 | 1.12M | let component_samples_needed = component.needed; | 505 | | | 506 | | // If image is interleaved iterate over scan components, | 507 | | // otherwise if it-s non-interleaved, these routines iterate in | 508 | | // trivial scanline order(Y,Cb,Cr) | 509 | | // | 510 | | // Turn the bounds into a compile time constant for a common special case. This | 511 | | // allows the compiler to unroll the loop and then do a bunch of interleaving. | 512 | | // | 513 | | // For PROGRESSIVE (non-interleaved), we iterate data units directly so | 514 | | // h_samp/v_samp loops run exactly once. | 515 | 1.12M | let v_step = | 516 | 1.12M | if SAMPLED && !PROGRESSIVE { 0..component.vertical_sample } else { 0..1 }; | 517 | | | 518 | 2.24M | for v_samp in v_step { | 519 | 1.12M | let h_step = | 520 | 1.12M | if SAMPLED && !PROGRESSIVE { 0..component.horizontal_sample } else { 0..1 }; | 521 | | | 522 | 2.24M | for h_samp in h_step { | 523 | 1.12M | let result = if component_samples_needed { | 524 | | // Fill the array with zeroes, decode_mcu_block expects | 525 | | // a zero based array. Clobber is in zig-zag order though. | 526 | | // Writing consecutive entries is basically free in terms | 527 | | // of memory throughput so we opt for a larger power of | 528 | | // two which lets the compiler turn this into a repeated | 529 | | // write of a zeroed vector register, which does not have | 530 | | // any branches, instead of a more difficult pattern where | 531 | | // we attempt to overwrite exactly one coefficient. | 532 | 1.12M | let clobber_len = if !clobber_more_than_4x4 { 32 } else { 64 }; | 533 | | | 534 | 1.12M | tmp[..clobber_len].fill(0); | 535 | | | 536 | 1.12M | stream.decode_mcu_block( | 537 | 1.12M | &mut self.stream, | 538 | 1.12M | dc_table, | 539 | 1.12M | ac_table, | 540 | 1.12M | qt_table, | 541 | 1.12M | tmp, | 542 | 1.12M | &mut component.dc_pred | 543 | | ) | 544 | | } else { | 545 | | // We do not touch tmp so there is no need to reset it. | 546 | 0 | stream.discard_mcu_block(&mut self.stream, dc_table, ac_table) | 547 | | }; | 548 | | | 549 | | // If an error occurs we can either propagate it | 550 | | // as an error or print it and call terminate. | 551 | | // | 552 | | // This allows even corrupt images to render something, | 553 | | // even if its bad, matching browsers. | 554 | | // | 555 | | // See example in https://github.com/etemesi254/zune-image/issues/293 | 556 | 1.12M | let len = if let Ok(len) = result { | 557 | 1.12M | len | 558 | | } else { | 559 | | // result.is_err() | 560 | 19 | return if self.options.strict_mode() { | 561 | 0 | Err(result.err().unwrap()) | 562 | | } else { | 563 | 19 | error!("{}", result.err().unwrap()); | 564 | 19 | Ok(McuContinuation::Terminate) | 565 | | }; | 566 | | }; | 567 | | | 568 | 1.12M | if component_samples_needed { | 569 | | // tmp was only written partially, note that len is in ZigZag order. | 570 | 1.12M | clobber_more_than_4x4 = len > 10; | 571 | | | 572 | 1.12M | let idct_position = if PROGRESSIVE { | 573 | | // For non-interleaved, j indexes data units directly | 574 | 1.12M | j * 8 | 575 | | } else { | 576 | | // derived from stb and rewritten for my tastes | 577 | 0 | let c2 = v_samp * 8; | 578 | 0 | let c3 = ((j * component.horizontal_sample) + h_samp) * 8; | 579 | | | 580 | 0 | component.width_stride * c2 + c3 | 581 | | }; | 582 | | | 583 | 1.12M | let idct_pos = channel.get_mut(idct_position..).unwrap(); | 584 | | | 585 | 1.12M | if len <= 1 { | 586 | 21.3k | (self.idct_1x1_func)(tmp, idct_pos, component.width_stride); | 587 | 1.10M | } else if len <= 10 { | 588 | 1.47k | (self.idct_4x4_func)(tmp, idct_pos, component.width_stride); | 589 | 1.10M | } else { | 590 | 1.10M | // call idct. | 591 | 1.10M | (self.idct_func)(tmp, idct_pos, component.width_stride); | 592 | 1.10M | } | 593 | 0 | } | 594 | | } | 595 | | } | 596 | | } | 597 | | | 598 | 285k | self.todo = self.todo.wrapping_sub(1); | 599 | | | 600 | 285k | if self.todo == 0 { | 601 | 345 | self.handle_rst_main(stream)?; | 602 | 345 | continue; | 603 | 285k | } | 604 | | | 605 | 285k | if stream.marker.is_some() && stream.bits_left == 0 { | 606 | 0 | break; | 607 | 285k | } | 608 | | } | 609 | | | 610 | 1.75k | self.check_stream_marker_after_mcu_width(stream) | 611 | 1.77k | } |
<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::inner_decode_mcu_width::<true, false> Line | Count | Source | 436 | 5.59k | fn inner_decode_mcu_width<const PROGRESSIVE: bool, const SAMPLED: bool>( | 437 | 5.59k | &mut self, mcu_width: usize, mcu_height: usize, tmp: &mut [i32; 64], | 438 | 5.59k | stream: &mut BitStream, progressive: &mut [Vec<i16>; 4] | 439 | 5.59k | ) -> Result<McuContinuation, DecodeErrors> { | 440 | 5.59k | let z_order = self.z_order; | 441 | 5.59k | let z_scans = &z_order[..usize::from(self.num_scans)]; | 442 | | | 443 | | // How much of the head of `tmp` was written by the last MCU decoding? We only check for | 444 | | // two different cases and not all possible outcomes as this is only used to optimize the | 445 | | // bytes written in `fill`. Since the clobber happens in UNZIGZAG order we'd be straddling | 446 | | // most cache lines anyways even if we did a partial write with the exact length of the | 447 | | // coefficient data which was written into `tmp`. | 448 | 5.59k | let mut clobber_more_than_4x4 = true; | 449 | | | 450 | | // For non-interleaved scans (PROGRESSIVE=true), each scan contains a single component | 451 | | // and we iterate over that component's actual data unit count, not the interleaved MCU | 452 | | // width multiplied by sampling factor. | 453 | 5.59k | let mut scan_du_width = if PROGRESSIVE { | 454 | 5.59k | let k = z_scans[0]; | 455 | 5.59k | let comp = &self.components[k]; | 456 | | // Calculate actual data units for this component: ceil(width / (8 * subsampling_ratio)) | 457 | 5.59k | (self.info.width as usize * comp.horizontal_sample + self.h_max * 8 - 1) | 458 | 5.59k | / (self.h_max * 8) | 459 | | } else { | 460 | 0 | mcu_width | 461 | | }; | 462 | | // In malformed scans that list multiple components, clamp to the smallest row capacity | 463 | | // to avoid writing past the row buffer. | 464 | 5.59k | if PROGRESSIVE && z_scans.len() > 1 { | 465 | 0 | let min_du = z_scans | 466 | 0 | .iter() | 467 | 0 | .map(|&k| self.components[k].width_stride / 8) | 468 | 0 | .min() | 469 | 0 | .unwrap_or(0); | 470 | 0 | scan_du_width = scan_du_width.min(min_du); | 471 | 5.59k | } | 472 | | | 473 | 294k | for j in 0..scan_du_width { | 474 | | // iterate over components | 475 | 588k | for &k in z_scans { | 476 | | // we made this loop body massive due to several different paths that depend on | 477 | | // static conditions. Note we (potentially) call into other functions so the | 478 | | // compiler will not unroll anything here anyways. The gains from separating | 479 | | // differently optimized loop bodies are much greater than a single additional jump | 480 | | // here. | 481 | 294k | let component = &mut self.components[k]; | 482 | | | 483 | 294k | let dc_table = self.dc_huffman_tables[component.dc_huff_table % MAX_COMPONENTS] | 484 | 294k | .as_ref() | 485 | 294k | .ok_or(DecodeErrors::FormatStatic("DC table not found"))?; | 486 | | | 487 | 294k | let ac_table = self.ac_huffman_tables[component.ac_huff_table % MAX_COMPONENTS] | 488 | 294k | .as_ref() | 489 | 294k | .ok_or(DecodeErrors::FormatStatic("AC table not found"))?; | 490 | | | 491 | 294k | let qt_table = &component.quantization_table; | 492 | 294k | let channel = if PROGRESSIVE { | 493 | 294k | let offset = | 494 | 294k | mcu_height * component.width_stride * 8 * component.vertical_sample; | 495 | | // Small stopgap for https://github.com/etemesi254/zune-image/issues/362 | 496 | 294k | if offset >= progressive[k].len(){ | 497 | 0 | return Err(DecodeErrors::FormatStatic("Would panic on slice iteration")) | 498 | 294k | } | 499 | 294k | &mut progressive[k][offset..] | 500 | | } else { | 501 | 0 | &mut component.raw_coeff | 502 | | }; | 503 | | | 504 | 294k | let component_samples_needed = component.needed; | 505 | | | 506 | | // If image is interleaved iterate over scan components, | 507 | | // otherwise if it-s non-interleaved, these routines iterate in | 508 | | // trivial scanline order(Y,Cb,Cr) | 509 | | // | 510 | | // Turn the bounds into a compile time constant for a common special case. This | 511 | | // allows the compiler to unroll the loop and then do a bunch of interleaving. | 512 | | // | 513 | | // For PROGRESSIVE (non-interleaved), we iterate data units directly so | 514 | | // h_samp/v_samp loops run exactly once. | 515 | 294k | let v_step = | 516 | 294k | if SAMPLED && !PROGRESSIVE { 0..component.vertical_sample } else { 0..1 }; | 517 | | | 518 | 588k | for v_samp in v_step { | 519 | 294k | let h_step = | 520 | 294k | if SAMPLED && !PROGRESSIVE { 0..component.horizontal_sample } else { 0..1 }; | 521 | | | 522 | 588k | for h_samp in h_step { | 523 | 294k | let result = if component_samples_needed { | 524 | | // Fill the array with zeroes, decode_mcu_block expects | 525 | | // a zero based array. Clobber is in zig-zag order though. | 526 | | // Writing consecutive entries is basically free in terms | 527 | | // of memory throughput so we opt for a larger power of | 528 | | // two which lets the compiler turn this into a repeated | 529 | | // write of a zeroed vector register, which does not have | 530 | | // any branches, instead of a more difficult pattern where | 531 | | // we attempt to overwrite exactly one coefficient. | 532 | 294k | let clobber_len = if !clobber_more_than_4x4 { 32 } else { 64 }; | 533 | | | 534 | 294k | tmp[..clobber_len].fill(0); | 535 | | | 536 | 294k | stream.decode_mcu_block( | 537 | 294k | &mut self.stream, | 538 | 294k | dc_table, | 539 | 294k | ac_table, | 540 | 294k | qt_table, | 541 | 294k | tmp, | 542 | 294k | &mut component.dc_pred | 543 | | ) | 544 | | } else { | 545 | | // We do not touch tmp so there is no need to reset it. | 546 | 0 | stream.discard_mcu_block(&mut self.stream, dc_table, ac_table) | 547 | | }; | 548 | | | 549 | | // If an error occurs we can either propagate it | 550 | | // as an error or print it and call terminate. | 551 | | // | 552 | | // This allows even corrupt images to render something, | 553 | | // even if its bad, matching browsers. | 554 | | // | 555 | | // See example in https://github.com/etemesi254/zune-image/issues/293 | 556 | 294k | let len = if let Ok(len) = result { | 557 | 294k | len | 558 | | } else { | 559 | | // result.is_err() | 560 | 81 | return if self.options.strict_mode() { | 561 | 0 | Err(result.err().unwrap()) | 562 | | } else { | 563 | 81 | error!("{}", result.err().unwrap()); | 564 | 81 | Ok(McuContinuation::Terminate) | 565 | | }; | 566 | | }; | 567 | | | 568 | 294k | if component_samples_needed { | 569 | | // tmp was only written partially, note that len is in ZigZag order. | 570 | 294k | clobber_more_than_4x4 = len > 10; | 571 | | | 572 | 294k | let idct_position = if PROGRESSIVE { | 573 | | // For non-interleaved, j indexes data units directly | 574 | 294k | j * 8 | 575 | | } else { | 576 | | // derived from stb and rewritten for my tastes | 577 | 0 | let c2 = v_samp * 8; | 578 | 0 | let c3 = ((j * component.horizontal_sample) + h_samp) * 8; | 579 | | | 580 | 0 | component.width_stride * c2 + c3 | 581 | | }; | 582 | | | 583 | 294k | let idct_pos = channel.get_mut(idct_position..).unwrap(); | 584 | | | 585 | 294k | if len <= 1 { | 586 | 259k | (self.idct_1x1_func)(tmp, idct_pos, component.width_stride); | 587 | 259k | } else if len <= 10 { | 588 | 5.19k | (self.idct_4x4_func)(tmp, idct_pos, component.width_stride); | 589 | 29.4k | } else { | 590 | 29.4k | // call idct. | 591 | 29.4k | (self.idct_func)(tmp, idct_pos, component.width_stride); | 592 | 29.4k | } | 593 | 0 | } | 594 | | } | 595 | | } | 596 | | } | 597 | | | 598 | 294k | self.todo = self.todo.wrapping_sub(1); | 599 | | | 600 | 294k | if self.todo == 0 { | 601 | 364 | self.handle_rst_main(stream)?; | 602 | 364 | continue; | 603 | 294k | } | 604 | | | 605 | 294k | if stream.marker.is_some() && stream.bits_left == 0 { | 606 | 0 | break; | 607 | 294k | } | 608 | | } | 609 | | | 610 | 5.51k | self.check_stream_marker_after_mcu_width(stream) | 611 | 5.59k | } |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::inner_decode_mcu_width::<_, _> |
612 | | |
613 | 37.6k | fn check_stream_marker_after_mcu_width( |
614 | 37.6k | &mut self, stream: &mut BitStream |
615 | 37.6k | ) -> Result<McuContinuation, DecodeErrors> { |
616 | | // After all interleaved components, that's an MCU |
617 | | // handle stream markers |
618 | | // |
619 | | // In some corrupt images, it may occur that header markers occur in the stream. |
620 | | // The spec EXPLICITLY FORBIDS this, specifically, in |
621 | | // routine F.2.2.5 it says |
622 | | // `The only valid marker which may occur within the Huffman coded data is the RSTm marker.` |
623 | | // |
624 | | // But libjpeg-turbo allows it because of some weird reason. so I'll also |
625 | | // allow it because of some weird reason. |
626 | 37.6k | if let Some(m) = stream.marker { |
627 | 11.0k | if m == Marker::EOI { |
628 | 189 | // acknowledge and ignore EOI marker. |
629 | 189 | stream.marker.take(); |
630 | 189 | trace!("Found EOI marker"); |
631 | 189 | // Google Introduced the Ultra-HD image format which is basically |
632 | 189 | // stitching two images into one container. |
633 | 189 | // They basically separate two images via a EOI and SOI marker |
634 | 189 | // so let's just ensure if we ever see EOI, we never read past that |
635 | 189 | // ever. |
636 | 189 | // https://github.com/google/libultrahdr |
637 | 189 | stream.seen_eoi = true; |
638 | 10.8k | } else if let Marker::RST(_) = m { |
639 | | //debug_assert_eq!(self.todo, 0); |
640 | 6.35k | if self.todo == 0 { |
641 | 0 | self.handle_rst(stream)?; |
642 | 6.35k | } |
643 | 4.53k | } else if let Marker::SOS = m { |
644 | 268 | self.parse_marker_inner(m)?; |
645 | 254 | stream.marker.take(); |
646 | 254 | stream.reset(); |
647 | | trace!("Found SOS marker"); |
648 | 254 | return Ok(McuContinuation::AnotherSos); |
649 | 4.27k | } else if matches!(m, Marker::DHT | Marker::DQT | Marker::DRI | Marker::COM) |
650 | 1.75k | || matches!(m, Marker::APP(_)) |
651 | | { |
652 | | // For non-interleaved images, setup markers can appear between scans. |
653 | | // Signal the caller to handle this marker and find the next SOS. |
654 | | // This keeps all marker parsing in the caller's loop. |
655 | 4.25k | stream.marker.take(); |
656 | | trace!("Found inter-scan marker {:?}", m); |
657 | 4.25k | return Ok(McuContinuation::InterScanMarker(m)); |
658 | | } else { |
659 | 12 | if self.options.strict_mode() { |
660 | 0 | return Err(DecodeErrors::Format(format!( |
661 | 0 | "Marker {m:?} found where not expected" |
662 | 0 | ))); |
663 | 12 | } |
664 | 12 | error!( |
665 | | "Marker `{:?}` Found within Huffman Stream, possibly corrupt jpeg", |
666 | | m |
667 | | ); |
668 | | |
669 | 12 | self.parse_marker_inner(m)?; |
670 | 10 | stream.marker.take(); |
671 | 10 | stream.reset(); |
672 | 10 | return Ok(McuContinuation::Terminate); |
673 | | } |
674 | 26.5k | } |
675 | | |
676 | 33.1k | Ok(McuContinuation::Ok) |
677 | 37.6k | } Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::check_stream_marker_after_mcu_width <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::check_stream_marker_after_mcu_width Line | Count | Source | 613 | 37.6k | fn check_stream_marker_after_mcu_width( | 614 | 37.6k | &mut self, stream: &mut BitStream | 615 | 37.6k | ) -> Result<McuContinuation, DecodeErrors> { | 616 | | // After all interleaved components, that's an MCU | 617 | | // handle stream markers | 618 | | // | 619 | | // In some corrupt images, it may occur that header markers occur in the stream. | 620 | | // The spec EXPLICITLY FORBIDS this, specifically, in | 621 | | // routine F.2.2.5 it says | 622 | | // `The only valid marker which may occur within the Huffman coded data is the RSTm marker.` | 623 | | // | 624 | | // But libjpeg-turbo allows it because of some weird reason. so I'll also | 625 | | // allow it because of some weird reason. | 626 | 37.6k | if let Some(m) = stream.marker { | 627 | 11.0k | if m == Marker::EOI { | 628 | 189 | // acknowledge and ignore EOI marker. | 629 | 189 | stream.marker.take(); | 630 | 189 | trace!("Found EOI marker"); | 631 | 189 | // Google Introduced the Ultra-HD image format which is basically | 632 | 189 | // stitching two images into one container. | 633 | 189 | // They basically separate two images via a EOI and SOI marker | 634 | 189 | // so let's just ensure if we ever see EOI, we never read past that | 635 | 189 | // ever. | 636 | 189 | // https://github.com/google/libultrahdr | 637 | 189 | stream.seen_eoi = true; | 638 | 10.8k | } else if let Marker::RST(_) = m { | 639 | | //debug_assert_eq!(self.todo, 0); | 640 | 6.35k | if self.todo == 0 { | 641 | 0 | self.handle_rst(stream)?; | 642 | 6.35k | } | 643 | 4.53k | } else if let Marker::SOS = m { | 644 | 268 | self.parse_marker_inner(m)?; | 645 | 254 | stream.marker.take(); | 646 | 254 | stream.reset(); | 647 | | trace!("Found SOS marker"); | 648 | 254 | return Ok(McuContinuation::AnotherSos); | 649 | 4.27k | } else if matches!(m, Marker::DHT | Marker::DQT | Marker::DRI | Marker::COM) | 650 | 1.75k | || matches!(m, Marker::APP(_)) | 651 | | { | 652 | | // For non-interleaved images, setup markers can appear between scans. | 653 | | // Signal the caller to handle this marker and find the next SOS. | 654 | | // This keeps all marker parsing in the caller's loop. | 655 | 4.25k | stream.marker.take(); | 656 | | trace!("Found inter-scan marker {:?}", m); | 657 | 4.25k | return Ok(McuContinuation::InterScanMarker(m)); | 658 | | } else { | 659 | 12 | if self.options.strict_mode() { | 660 | 0 | return Err(DecodeErrors::Format(format!( | 661 | 0 | "Marker {m:?} found where not expected" | 662 | 0 | ))); | 663 | 12 | } | 664 | 12 | error!( | 665 | | "Marker `{:?}` Found within Huffman Stream, possibly corrupt jpeg", | 666 | | m | 667 | | ); | 668 | | | 669 | 12 | self.parse_marker_inner(m)?; | 670 | 10 | stream.marker.take(); | 671 | 10 | stream.reset(); | 672 | 10 | return Ok(McuContinuation::Terminate); | 673 | | } | 674 | 26.5k | } | 675 | | | 676 | 33.1k | Ok(McuContinuation::Ok) | 677 | 37.6k | } |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::check_stream_marker_after_mcu_width |
678 | | |
679 | | /// Scan for the next SOS marker, parsing setup markers along the way. |
680 | | /// |
681 | | /// This is the unified marker scanning function used after encountering an |
682 | | /// inter-scan marker. It handles DHT, DQT, DRI, COM, and APP markers that |
683 | | /// can appear between scans in non-interleaved images. |
684 | | /// |
685 | | /// # Arguments |
686 | | /// * `first_marker` - The first marker that was already detected (not yet parsed) |
687 | | /// * `stream` - The bitstream state |
688 | | /// |
689 | | /// # Returns |
690 | | /// * `Ok(true)` - Found SOS, ready to continue decoding |
691 | | /// * `Ok(false)` - Found EOI, decoding complete |
692 | | /// * `Err(_)` - Error (too many markers, unexpected marker in strict mode, etc.) |
693 | 4.25k | fn advance_to_next_sos( |
694 | 4.25k | &mut self, |
695 | 4.25k | first_marker: Marker, |
696 | 4.25k | stream: &mut BitStream |
697 | 4.25k | ) -> Result<bool, DecodeErrors> { |
698 | | // Limit iterations to prevent DoS from malicious files. |
699 | | const MAX_INTER_SCAN_MARKERS: usize = 64; |
700 | | |
701 | | // Parse the first marker that triggered this call |
702 | 4.25k | self.parse_marker_inner(first_marker)?; |
703 | 4.23k | stream.reset(); |
704 | | |
705 | 18.7k | for _ in 0..MAX_INTER_SCAN_MARKERS { |
706 | 18.7k | let marker = get_marker(&mut self.stream, stream)?; |
707 | | |
708 | 18.6k | match marker { |
709 | | Marker::SOS => { |
710 | 4.15k | self.parse_marker_inner(Marker::SOS)?; |
711 | 4.14k | stream.reset(); |
712 | | trace!("Found SOS marker, continuing decode"); |
713 | 4.14k | return Ok(true); |
714 | | } |
715 | | Marker::EOI => { |
716 | 14 | stream.seen_eoi = true; |
717 | | trace!("Found EOI marker"); |
718 | 14 | return Ok(false); |
719 | | } |
720 | | Marker::DHT | Marker::DQT | Marker::DRI | Marker::COM => { |
721 | | trace!("Parsing inter-scan marker {:?}", marker); |
722 | 3.81k | self.parse_marker_inner(marker)?; |
723 | | } |
724 | | Marker::APP(_) => { |
725 | | trace!("Parsing inter-scan APP marker {:?}", marker); |
726 | 5.79k | self.parse_marker_inner(marker)?; |
727 | | } |
728 | 4.92k | other => { |
729 | 4.92k | if self.options.strict_mode() { |
730 | 0 | return Err(DecodeErrors::Format(format!( |
731 | 0 | "Unexpected marker {:?} while scanning for SOS between scans", |
732 | 0 | other |
733 | 0 | ))); |
734 | 4.92k | } |
735 | | // Non-strict: skip unknown marker |
736 | 4.92k | warn!("Skipping unexpected marker {:?} between scans", other); |
737 | 4.92k | let length = self.stream.get_u16_be_err()?; |
738 | 4.92k | if length >= 2 { |
739 | 4.62k | self.stream.skip((length - 2) as usize)?; |
740 | 301 | } |
741 | | } |
742 | | } |
743 | | } |
744 | | |
745 | 1 | Err(DecodeErrors::FormatStatic( |
746 | 1 | "Too many markers between scans (exceeded limit of 64)" |
747 | 1 | )) |
748 | 4.25k | } Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::advance_to_next_sos <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::advance_to_next_sos Line | Count | Source | 693 | 4.25k | fn advance_to_next_sos( | 694 | 4.25k | &mut self, | 695 | 4.25k | first_marker: Marker, | 696 | 4.25k | stream: &mut BitStream | 697 | 4.25k | ) -> Result<bool, DecodeErrors> { | 698 | | // Limit iterations to prevent DoS from malicious files. | 699 | | const MAX_INTER_SCAN_MARKERS: usize = 64; | 700 | | | 701 | | // Parse the first marker that triggered this call | 702 | 4.25k | self.parse_marker_inner(first_marker)?; | 703 | 4.23k | stream.reset(); | 704 | | | 705 | 18.7k | for _ in 0..MAX_INTER_SCAN_MARKERS { | 706 | 18.7k | let marker = get_marker(&mut self.stream, stream)?; | 707 | | | 708 | 18.6k | match marker { | 709 | | Marker::SOS => { | 710 | 4.15k | self.parse_marker_inner(Marker::SOS)?; | 711 | 4.14k | stream.reset(); | 712 | | trace!("Found SOS marker, continuing decode"); | 713 | 4.14k | return Ok(true); | 714 | | } | 715 | | Marker::EOI => { | 716 | 14 | stream.seen_eoi = true; | 717 | | trace!("Found EOI marker"); | 718 | 14 | return Ok(false); | 719 | | } | 720 | | Marker::DHT | Marker::DQT | Marker::DRI | Marker::COM => { | 721 | | trace!("Parsing inter-scan marker {:?}", marker); | 722 | 3.81k | self.parse_marker_inner(marker)?; | 723 | | } | 724 | | Marker::APP(_) => { | 725 | | trace!("Parsing inter-scan APP marker {:?}", marker); | 726 | 5.79k | self.parse_marker_inner(marker)?; | 727 | | } | 728 | 4.92k | other => { | 729 | 4.92k | if self.options.strict_mode() { | 730 | 0 | return Err(DecodeErrors::Format(format!( | 731 | 0 | "Unexpected marker {:?} while scanning for SOS between scans", | 732 | 0 | other | 733 | 0 | ))); | 734 | 4.92k | } | 735 | | // Non-strict: skip unknown marker | 736 | 4.92k | warn!("Skipping unexpected marker {:?} between scans", other); | 737 | 4.92k | let length = self.stream.get_u16_be_err()?; | 738 | 4.92k | if length >= 2 { | 739 | 4.62k | self.stream.skip((length - 2) as usize)?; | 740 | 301 | } | 741 | | } | 742 | | } | 743 | | } | 744 | | | 745 | 1 | Err(DecodeErrors::FormatStatic( | 746 | 1 | "Too many markers between scans (exceeded limit of 64)" | 747 | 1 | )) | 748 | 4.25k | } |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::advance_to_next_sos |
749 | | |
750 | | // handle RST markers. |
751 | | // No-op if not using restarts |
752 | | // this routine is shared with mcu_prog |
753 | | #[cold] |
754 | 34.2k | pub(crate) fn handle_rst(&mut self, stream: &mut BitStream) -> Result<(), DecodeErrors> { |
755 | 34.2k | self.todo = self.restart_interval; |
756 | | |
757 | 34.2k | if let Some(marker) = stream.marker { |
758 | | // Found a marker |
759 | | // Read stream and see what marker is stored there |
760 | 30.9k | match marker { |
761 | | Marker::RST(_) => { |
762 | | // reset stream |
763 | 3.05k | stream.reset(); |
764 | | // Initialize dc predictions to zero for all components |
765 | 3.05k | self.components.iter_mut().for_each(|x| x.dc_pred = 0); |
766 | | // Start iterating again. from position. |
767 | | } |
768 | 708 | Marker::EOI => { |
769 | 708 | // silent pass |
770 | 708 | } |
771 | | // Valid markers that can appear between scans at a restart boundary |
772 | | // (restart interval aligns with end of scan). Leave for caller. |
773 | | Marker::SOS | Marker::DHT | Marker::DQT | Marker::DRI | Marker::COM |
774 | 17.4k | | Marker::APP(_) => {} |
775 | | _ => { |
776 | 9.70k | if self.options.strict_mode() { |
777 | 0 | return Err(DecodeErrors::MCUError(format!( |
778 | 0 | "Unexpected marker {marker:?} at restart boundary" |
779 | 0 | ))); |
780 | 9.70k | } |
781 | 9.70k | warn!("Unexpected marker {:?} at restart boundary", marker); |
782 | | } |
783 | | } |
784 | 3.37k | } |
785 | 34.2k | Ok(()) |
786 | 34.2k | } Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::handle_rst <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::handle_rst Line | Count | Source | 754 | 34.2k | pub(crate) fn handle_rst(&mut self, stream: &mut BitStream) -> Result<(), DecodeErrors> { | 755 | 34.2k | self.todo = self.restart_interval; | 756 | | | 757 | 34.2k | if let Some(marker) = stream.marker { | 758 | | // Found a marker | 759 | | // Read stream and see what marker is stored there | 760 | 30.9k | match marker { | 761 | | Marker::RST(_) => { | 762 | | // reset stream | 763 | 3.05k | stream.reset(); | 764 | | // Initialize dc predictions to zero for all components | 765 | 3.05k | self.components.iter_mut().for_each(|x| x.dc_pred = 0); | 766 | | // Start iterating again. from position. | 767 | | } | 768 | 708 | Marker::EOI => { | 769 | 708 | // silent pass | 770 | 708 | } | 771 | | // Valid markers that can appear between scans at a restart boundary | 772 | | // (restart interval aligns with end of scan). Leave for caller. | 773 | | Marker::SOS | Marker::DHT | Marker::DQT | Marker::DRI | Marker::COM | 774 | 17.4k | | Marker::APP(_) => {} | 775 | | _ => { | 776 | 9.70k | if self.options.strict_mode() { | 777 | 0 | return Err(DecodeErrors::MCUError(format!( | 778 | 0 | "Unexpected marker {marker:?} at restart boundary" | 779 | 0 | ))); | 780 | 9.70k | } | 781 | 9.70k | warn!("Unexpected marker {:?} at restart boundary", marker); | 782 | | } | 783 | | } | 784 | 3.37k | } | 785 | 34.2k | Ok(()) | 786 | 34.2k | } |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::handle_rst |
787 | | #[allow(clippy::too_many_lines, clippy::too_many_arguments)] |
788 | 132k | pub(crate) fn post_process( |
789 | 132k | &mut self, pixels: &mut [u8], i: usize, mcu_height: usize, width: usize, |
790 | 132k | padded_width: usize, pixels_written: &mut usize, upsampler_scratch_space: &mut [i16] |
791 | 132k | ) -> Result<(), DecodeErrors> { |
792 | 132k | let out_colorspace_components = self.options.jpeg_get_out_colorspace().num_components(); |
793 | | |
794 | 132k | let mut px = *pixels_written; |
795 | | // indicates whether image is vertically up-sampled |
796 | 132k | let is_vertically_sampled = self |
797 | 132k | .components |
798 | 132k | .iter() |
799 | 197k | .any(|c| c.sample_ratio == SampleRatios::HV || c.sample_ratio == SampleRatios::V); Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::post_process::{closure#0}<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::post_process::{closure#0}Line | Count | Source | 799 | 197k | .any(|c| c.sample_ratio == SampleRatios::HV || c.sample_ratio == SampleRatios::V); |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::post_process::{closure#0} |
800 | | |
801 | 132k | let mut comp_len = self.components.len(); |
802 | | |
803 | | // If we are moving from YCbCr -> Luma, we do not allocate storage for other components, so we |
804 | | // will panic when we are trying to read samples, so for that case, |
805 | | // hardcode it so that we don't panic when doing |
806 | | // *samp = &samples[j][pos * padded_width..(pos + 1) * padded_width] |
807 | 132k | if out_colorspace_components < comp_len && self.options.jpeg_get_out_colorspace() == Luma { |
808 | 0 | comp_len = out_colorspace_components; |
809 | 132k | } |
810 | 132k | let mut color_conv_function = |
811 | 220k | |num_iters: usize, samples: [&[i16]; 4]| -> Result<(), DecodeErrors> { |
812 | 1.99M | for (pos, output) in pixels[px..] |
813 | 220k | .chunks_exact_mut(width * out_colorspace_components) |
814 | 220k | .take(num_iters) |
815 | 220k | .enumerate() |
816 | | { |
817 | 1.99M | let mut raw_samples: [&[i16]; 4] = [&[], &[], &[], &[]]; |
818 | | |
819 | | // iterate over each line, since color-convert needs only |
820 | | // one line |
821 | 7.48M | for (j, samp) in raw_samples.iter_mut().enumerate().take(comp_len) { |
822 | 7.48M | let temp = &samples[j].get(pos * padded_width..(pos + 1) * padded_width); |
823 | 7.48M | if temp.is_none() { |
824 | 5 | return Err(DecodeErrors::FormatStatic("Missing samples")); |
825 | 7.48M | } |
826 | 7.48M | *samp = temp.unwrap(); |
827 | | } |
828 | 1.99M | color_convert( |
829 | 1.99M | &raw_samples, |
830 | 1.99M | self.color_convert_16, |
831 | 1.99M | self.input_colorspace, |
832 | 1.99M | self.options.jpeg_get_out_colorspace(), |
833 | 1.99M | output, |
834 | 1.99M | width, |
835 | 1.99M | padded_width |
836 | 43 | )?; |
837 | 1.99M | px += width * out_colorspace_components; |
838 | | } |
839 | 220k | Ok(()) |
840 | 220k | }; Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::post_process::{closure#1}<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::post_process::{closure#1}Line | Count | Source | 811 | 220k | |num_iters: usize, samples: [&[i16]; 4]| -> Result<(), DecodeErrors> { | 812 | 1.99M | for (pos, output) in pixels[px..] | 813 | 220k | .chunks_exact_mut(width * out_colorspace_components) | 814 | 220k | .take(num_iters) | 815 | 220k | .enumerate() | 816 | | { | 817 | 1.99M | let mut raw_samples: [&[i16]; 4] = [&[], &[], &[], &[]]; | 818 | | | 819 | | // iterate over each line, since color-convert needs only | 820 | | // one line | 821 | 7.48M | for (j, samp) in raw_samples.iter_mut().enumerate().take(comp_len) { | 822 | 7.48M | let temp = &samples[j].get(pos * padded_width..(pos + 1) * padded_width); | 823 | 7.48M | if temp.is_none() { | 824 | 5 | return Err(DecodeErrors::FormatStatic("Missing samples")); | 825 | 7.48M | } | 826 | 7.48M | *samp = temp.unwrap(); | 827 | | } | 828 | 1.99M | color_convert( | 829 | 1.99M | &raw_samples, | 830 | 1.99M | self.color_convert_16, | 831 | 1.99M | self.input_colorspace, | 832 | 1.99M | self.options.jpeg_get_out_colorspace(), | 833 | 1.99M | output, | 834 | 1.99M | width, | 835 | 1.99M | padded_width | 836 | 43 | )?; | 837 | 1.99M | px += width * out_colorspace_components; | 838 | | } | 839 | 220k | Ok(()) | 840 | 220k | }; |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::post_process::{closure#1} |
841 | | |
842 | 132k | let comps = &mut self.components[..]; |
843 | | |
844 | 132k | if self.is_interleaved && self.options.jpeg_get_out_colorspace() != ColorSpace::Luma { |
845 | 440k | for comp in comps.iter_mut() { |
846 | 440k | upsample( |
847 | 440k | comp, |
848 | 440k | mcu_height, |
849 | 440k | i, |
850 | 440k | upsampler_scratch_space, |
851 | 440k | is_vertically_sampled |
852 | 0 | )?; |
853 | | } |
854 | | |
855 | 111k | if is_vertically_sampled { |
856 | 89.9k | if i > 0 { |
857 | | // write the last line, it wasn't up-sampled as we didn't have row_down |
858 | | // yet |
859 | 88.2k | let mut samples: [&[i16]; 4] = [&[], &[], &[], &[]]; |
860 | | |
861 | 353k | for (samp, component) in samples.iter_mut().zip(comps.iter()) { |
862 | 353k | *samp = &component.first_row_upsample_dest; |
863 | 353k | } |
864 | | |
865 | | // ensure length matches for all samples |
866 | 88.2k | let _first_len = samples[0].len(); |
867 | | |
868 | | // This was a good check, but can be caused to panic, esp on invalid/corrupt images. |
869 | | // See one in issue https://github.com/etemesi254/zune-image/issues/262, so for now |
870 | | // we just ignore and generate invalid images at the end. |
871 | | |
872 | | // |
873 | | // |
874 | | // for samp in samples.iter().take(comp_len) { |
875 | | // assert_eq!(first_len, samp.len()); |
876 | | // } |
877 | 88.2k | let num_iters = self.coeff * self.v_max; |
878 | | |
879 | 88.2k | color_conv_function(num_iters, samples)?; |
880 | 1.64k | } |
881 | | |
882 | | // After up-sampling the last row, save any row that can be used for |
883 | | // a later up-sampling, |
884 | | // |
885 | | // E.g the Y sample is not sampled but we haven't finished upsampling the last row of |
886 | | // the previous mcu, since we don't have the down row, so save it |
887 | 359k | for component in comps.iter_mut() { |
888 | 359k | if component.sample_ratio != SampleRatios::H { |
889 | 359k | // We don't care about H sampling factors, since it's copied in the workers function |
890 | 359k | |
891 | 359k | // copy last row to be used for the next color conversion |
892 | 359k | let size = component.vertical_sample |
893 | 359k | * component.width_stride |
894 | 359k | * component.sample_ratio.sample(); |
895 | 359k | |
896 | 359k | let last_bytes = |
897 | 359k | component.raw_coeff.rchunks_exact_mut(size).next().unwrap(); |
898 | 359k | |
899 | 359k | component |
900 | 359k | .first_row_upsample_dest |
901 | 359k | .copy_from_slice(last_bytes); |
902 | 359k | } |
903 | | } |
904 | 22.0k | } |
905 | | |
906 | 111k | let mut samples: [&[i16]; 4] = [&[], &[], &[], &[]]; |
907 | | |
908 | 440k | for (samp, component) in samples.iter_mut().zip(comps.iter()) { |
909 | 440k | *samp = if component.sample_ratio == SampleRatios::None { |
910 | 110k | &component.raw_coeff |
911 | | } else { |
912 | 330k | &component.upsample_dest |
913 | | }; |
914 | | } |
915 | | |
916 | | // we either do 7 or 8 MCU's depending on the state, this only applies to |
917 | | // vertically sampled images |
918 | | // |
919 | | // for rows up until the last MCU, we do not upsample the last stride of the MCU |
920 | | // which means that the number of iterations should take that into account is one less the |
921 | | // up-sampled size |
922 | | // |
923 | | // For the last MCU, we upsample the last stride, meaning that if we hit the last MCU, we |
924 | | // should sample full raw coeffs |
925 | 111k | let is_last_considered = is_vertically_sampled && (i != mcu_height.saturating_sub(1)); |
926 | | |
927 | 111k | let num_iters = (8 - usize::from(is_last_considered)) * self.coeff * self.v_max; |
928 | | |
929 | 111k | color_conv_function(num_iters, samples)?; |
930 | | } else { |
931 | 20.1k | let mut channels_ref: [&[i16]; MAX_COMPONENTS] = [&[]; MAX_COMPONENTS]; |
932 | | |
933 | 20.1k | self.components |
934 | 20.1k | .iter() |
935 | 20.1k | .enumerate() |
936 | 25.2k | .for_each(|(pos, x)| channels_ref[pos] = &x.raw_coeff); Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::post_process::{closure#2}<zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::post_process::{closure#2}Line | Count | Source | 936 | 25.2k | .for_each(|(pos, x)| channels_ref[pos] = &x.raw_coeff); |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::post_process::{closure#2} |
937 | | |
938 | 20.1k | if let SampleRatios::Generic(_, v) = self.info.sample_ratio { |
939 | 0 | color_conv_function(8 * v * self.coeff, channels_ref)?; |
940 | | } else { |
941 | 20.1k | color_conv_function(8 * self.coeff, channels_ref)?; |
942 | | } |
943 | | } |
944 | | |
945 | 131k | *pixels_written = px; |
946 | 131k | Ok(()) |
947 | 132k | } Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<std::io::cursor::Cursor<&[u8]>>>::post_process <zune_jpeg::decoder::JpegDecoder<zune_core::bytestream::reader::no_std_readers::ZCursor<alloc::vec::Vec<u8>>>>::post_process Line | Count | Source | 788 | 132k | pub(crate) fn post_process( | 789 | 132k | &mut self, pixels: &mut [u8], i: usize, mcu_height: usize, width: usize, | 790 | 132k | padded_width: usize, pixels_written: &mut usize, upsampler_scratch_space: &mut [i16] | 791 | 132k | ) -> Result<(), DecodeErrors> { | 792 | 132k | let out_colorspace_components = self.options.jpeg_get_out_colorspace().num_components(); | 793 | | | 794 | 132k | let mut px = *pixels_written; | 795 | | // indicates whether image is vertically up-sampled | 796 | 132k | let is_vertically_sampled = self | 797 | 132k | .components | 798 | 132k | .iter() | 799 | 132k | .any(|c| c.sample_ratio == SampleRatios::HV || c.sample_ratio == SampleRatios::V); | 800 | | | 801 | 132k | let mut comp_len = self.components.len(); | 802 | | | 803 | | // If we are moving from YCbCr -> Luma, we do not allocate storage for other components, so we | 804 | | // will panic when we are trying to read samples, so for that case, | 805 | | // hardcode it so that we don't panic when doing | 806 | | // *samp = &samples[j][pos * padded_width..(pos + 1) * padded_width] | 807 | 132k | if out_colorspace_components < comp_len && self.options.jpeg_get_out_colorspace() == Luma { | 808 | 0 | comp_len = out_colorspace_components; | 809 | 132k | } | 810 | 132k | let mut color_conv_function = | 811 | | |num_iters: usize, samples: [&[i16]; 4]| -> Result<(), DecodeErrors> { | 812 | | for (pos, output) in pixels[px..] | 813 | | .chunks_exact_mut(width * out_colorspace_components) | 814 | | .take(num_iters) | 815 | | .enumerate() | 816 | | { | 817 | | let mut raw_samples: [&[i16]; 4] = [&[], &[], &[], &[]]; | 818 | | | 819 | | // iterate over each line, since color-convert needs only | 820 | | // one line | 821 | | for (j, samp) in raw_samples.iter_mut().enumerate().take(comp_len) { | 822 | | let temp = &samples[j].get(pos * padded_width..(pos + 1) * padded_width); | 823 | | if temp.is_none() { | 824 | | return Err(DecodeErrors::FormatStatic("Missing samples")); | 825 | | } | 826 | | *samp = temp.unwrap(); | 827 | | } | 828 | | color_convert( | 829 | | &raw_samples, | 830 | | self.color_convert_16, | 831 | | self.input_colorspace, | 832 | | self.options.jpeg_get_out_colorspace(), | 833 | | output, | 834 | | width, | 835 | | padded_width | 836 | | )?; | 837 | | px += width * out_colorspace_components; | 838 | | } | 839 | | Ok(()) | 840 | | }; | 841 | | | 842 | 132k | let comps = &mut self.components[..]; | 843 | | | 844 | 132k | if self.is_interleaved && self.options.jpeg_get_out_colorspace() != ColorSpace::Luma { | 845 | 440k | for comp in comps.iter_mut() { | 846 | 440k | upsample( | 847 | 440k | comp, | 848 | 440k | mcu_height, | 849 | 440k | i, | 850 | 440k | upsampler_scratch_space, | 851 | 440k | is_vertically_sampled | 852 | 0 | )?; | 853 | | } | 854 | | | 855 | 111k | if is_vertically_sampled { | 856 | 89.9k | if i > 0 { | 857 | | // write the last line, it wasn't up-sampled as we didn't have row_down | 858 | | // yet | 859 | 88.2k | let mut samples: [&[i16]; 4] = [&[], &[], &[], &[]]; | 860 | | | 861 | 353k | for (samp, component) in samples.iter_mut().zip(comps.iter()) { | 862 | 353k | *samp = &component.first_row_upsample_dest; | 863 | 353k | } | 864 | | | 865 | | // ensure length matches for all samples | 866 | 88.2k | let _first_len = samples[0].len(); | 867 | | | 868 | | // This was a good check, but can be caused to panic, esp on invalid/corrupt images. | 869 | | // See one in issue https://github.com/etemesi254/zune-image/issues/262, so for now | 870 | | // we just ignore and generate invalid images at the end. | 871 | | | 872 | | // | 873 | | // | 874 | | // for samp in samples.iter().take(comp_len) { | 875 | | // assert_eq!(first_len, samp.len()); | 876 | | // } | 877 | 88.2k | let num_iters = self.coeff * self.v_max; | 878 | | | 879 | 88.2k | color_conv_function(num_iters, samples)?; | 880 | 1.64k | } | 881 | | | 882 | | // After up-sampling the last row, save any row that can be used for | 883 | | // a later up-sampling, | 884 | | // | 885 | | // E.g the Y sample is not sampled but we haven't finished upsampling the last row of | 886 | | // the previous mcu, since we don't have the down row, so save it | 887 | 359k | for component in comps.iter_mut() { | 888 | 359k | if component.sample_ratio != SampleRatios::H { | 889 | 359k | // We don't care about H sampling factors, since it's copied in the workers function | 890 | 359k | | 891 | 359k | // copy last row to be used for the next color conversion | 892 | 359k | let size = component.vertical_sample | 893 | 359k | * component.width_stride | 894 | 359k | * component.sample_ratio.sample(); | 895 | 359k | | 896 | 359k | let last_bytes = | 897 | 359k | component.raw_coeff.rchunks_exact_mut(size).next().unwrap(); | 898 | 359k | | 899 | 359k | component | 900 | 359k | .first_row_upsample_dest | 901 | 359k | .copy_from_slice(last_bytes); | 902 | 359k | } | 903 | | } | 904 | 22.0k | } | 905 | | | 906 | 111k | let mut samples: [&[i16]; 4] = [&[], &[], &[], &[]]; | 907 | | | 908 | 440k | for (samp, component) in samples.iter_mut().zip(comps.iter()) { | 909 | 440k | *samp = if component.sample_ratio == SampleRatios::None { | 910 | 110k | &component.raw_coeff | 911 | | } else { | 912 | 330k | &component.upsample_dest | 913 | | }; | 914 | | } | 915 | | | 916 | | // we either do 7 or 8 MCU's depending on the state, this only applies to | 917 | | // vertically sampled images | 918 | | // | 919 | | // for rows up until the last MCU, we do not upsample the last stride of the MCU | 920 | | // which means that the number of iterations should take that into account is one less the | 921 | | // up-sampled size | 922 | | // | 923 | | // For the last MCU, we upsample the last stride, meaning that if we hit the last MCU, we | 924 | | // should sample full raw coeffs | 925 | 111k | let is_last_considered = is_vertically_sampled && (i != mcu_height.saturating_sub(1)); | 926 | | | 927 | 111k | let num_iters = (8 - usize::from(is_last_considered)) * self.coeff * self.v_max; | 928 | | | 929 | 111k | color_conv_function(num_iters, samples)?; | 930 | | } else { | 931 | 20.1k | let mut channels_ref: [&[i16]; MAX_COMPONENTS] = [&[]; MAX_COMPONENTS]; | 932 | | | 933 | 20.1k | self.components | 934 | 20.1k | .iter() | 935 | 20.1k | .enumerate() | 936 | 20.1k | .for_each(|(pos, x)| channels_ref[pos] = &x.raw_coeff); | 937 | | | 938 | 20.1k | if let SampleRatios::Generic(_, v) = self.info.sample_ratio { | 939 | 0 | color_conv_function(8 * v * self.coeff, channels_ref)?; | 940 | | } else { | 941 | 20.1k | color_conv_function(8 * self.coeff, channels_ref)?; | 942 | | } | 943 | | } | 944 | | | 945 | 131k | *pixels_written = px; | 946 | 131k | Ok(()) | 947 | 132k | } |
Unexecuted instantiation: <zune_jpeg::decoder::JpegDecoder<_>>::post_process |
948 | | } |
949 | | |
950 | | enum McuContinuation { |
951 | | Ok, |
952 | | AnotherSos, |
953 | | /// Found an inter-scan marker (DHT/DQT/DRI/COM/APP) that needs handling. |
954 | | /// The caller should parse it and scan for the next SOS. |
955 | | InterScanMarker(Marker), |
956 | | Terminate |
957 | | } |