Coverage Report

Created: 2026-09-14 07:37

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libjxl/lib/jxl/compressed_dc.cc
Line
Count
Source
1
// Copyright (c) the JPEG XL Project Authors. All rights reserved.
2
//
3
// Use of this source code is governed by a BSD-style
4
// license that can be found in the LICENSE file.
5
6
#include "lib/jxl/compressed_dc.h"
7
8
#include <jxl/memory_manager.h>
9
10
#include <algorithm>
11
#include <cstdint>
12
#include <cstdlib>
13
#include <cstring>
14
#include <vector>
15
16
#include "lib/jxl/ac_context.h"
17
#include "lib/jxl/frame_header.h"
18
#include "lib/jxl/modular/modular_image.h"
19
20
#undef HWY_TARGET_INCLUDE
21
#define HWY_TARGET_INCLUDE "lib/jxl/compressed_dc.cc"
22
#include <hwy/foreach_target.h>
23
#include <hwy/highway.h>
24
25
#include "lib/jxl/base/compiler_specific.h"
26
#include "lib/jxl/base/data_parallel.h"
27
#include "lib/jxl/base/rect.h"
28
#include "lib/jxl/base/status.h"
29
#include "lib/jxl/image.h"
30
HWY_BEFORE_NAMESPACE();
31
namespace jxl {
32
namespace HWY_NAMESPACE {
33
34
using D = HWY_FULL(float);
35
using DScalar = HWY_CAPPED(float, 1);
36
37
// These templates are not found via ADL.
38
using hwy::HWY_NAMESPACE::Abs;
39
using hwy::HWY_NAMESPACE::Add;
40
using hwy::HWY_NAMESPACE::Div;
41
using hwy::HWY_NAMESPACE::Max;
42
using hwy::HWY_NAMESPACE::Mul;
43
using hwy::HWY_NAMESPACE::MulAdd;
44
using hwy::HWY_NAMESPACE::Rebind;
45
using hwy::HWY_NAMESPACE::Sub;
46
using hwy::HWY_NAMESPACE::Vec;
47
using hwy::HWY_NAMESPACE::ZeroIfNegative;
48
49
// TODO(veluca): optimize constants.
50
const float w1 = 0.20345139757231578f;
51
const float w2 = 0.0334829185968739f;
52
const float w0 = 1.0f - 4.0f * (w1 + w2);
53
54
template <class V>
55
12.0M
V MaxWorkaround(V a, V b) {
56
#if (HWY_TARGET == HWY_AVX3) && HWY_COMPILER_CLANG <= 800
57
  // Prevents "Do not know how to split the result of this operator" error
58
  return IfThenElse(a > b, a, b);
59
#else
60
12.0M
  return Max(a, b);
61
12.0M
#endif
62
12.0M
}
Unexecuted instantiation: hwy::N_SSE4::Vec128<float, 1ul> jxl::N_SSE4::MaxWorkaround<hwy::N_SSE4::Vec128<float, 1ul> >(hwy::N_SSE4::Vec128<float, 1ul>, hwy::N_SSE4::Vec128<float, 1ul>)
Unexecuted instantiation: hwy::N_SSE4::Vec128<float, 4ul> jxl::N_SSE4::MaxWorkaround<hwy::N_SSE4::Vec128<float, 4ul> >(hwy::N_SSE4::Vec128<float, 4ul>, hwy::N_SSE4::Vec128<float, 4ul>)
hwy::N_AVX2::Vec128<float, 1ul> jxl::N_AVX2::MaxWorkaround<hwy::N_AVX2::Vec128<float, 1ul> >(hwy::N_AVX2::Vec128<float, 1ul>, hwy::N_AVX2::Vec128<float, 1ul>)
Line
Count
Source
55
9.70M
V MaxWorkaround(V a, V b) {
56
#if (HWY_TARGET == HWY_AVX3) && HWY_COMPILER_CLANG <= 800
57
  // Prevents "Do not know how to split the result of this operator" error
58
  return IfThenElse(a > b, a, b);
59
#else
60
9.70M
  return Max(a, b);
61
9.70M
#endif
62
9.70M
}
hwy::N_AVX2::Vec256<float> jxl::N_AVX2::MaxWorkaround<hwy::N_AVX2::Vec256<float> >(hwy::N_AVX2::Vec256<float>, hwy::N_AVX2::Vec256<float>)
Line
Count
Source
55
2.36M
V MaxWorkaround(V a, V b) {
56
#if (HWY_TARGET == HWY_AVX3) && HWY_COMPILER_CLANG <= 800
57
  // Prevents "Do not know how to split the result of this operator" error
58
  return IfThenElse(a > b, a, b);
59
#else
60
2.36M
  return Max(a, b);
61
2.36M
#endif
62
2.36M
}
Unexecuted instantiation: hwy::N_AVX3::Vec128<float, 1ul> jxl::N_AVX3::MaxWorkaround<hwy::N_AVX3::Vec128<float, 1ul> >(hwy::N_AVX3::Vec128<float, 1ul>, hwy::N_AVX3::Vec128<float, 1ul>)
Unexecuted instantiation: hwy::N_AVX3::Vec512<float> jxl::N_AVX3::MaxWorkaround<hwy::N_AVX3::Vec512<float> >(hwy::N_AVX3::Vec512<float>, hwy::N_AVX3::Vec512<float>)
Unexecuted instantiation: hwy::N_AVX3_ZEN4::Vec128<float, 1ul> jxl::N_AVX3_ZEN4::MaxWorkaround<hwy::N_AVX3_ZEN4::Vec128<float, 1ul> >(hwy::N_AVX3_ZEN4::Vec128<float, 1ul>, hwy::N_AVX3_ZEN4::Vec128<float, 1ul>)
Unexecuted instantiation: hwy::N_AVX3_ZEN4::Vec512<float> jxl::N_AVX3_ZEN4::MaxWorkaround<hwy::N_AVX3_ZEN4::Vec512<float> >(hwy::N_AVX3_ZEN4::Vec512<float>, hwy::N_AVX3_ZEN4::Vec512<float>)
Unexecuted instantiation: hwy::N_AVX3_SPR::Vec128<float, 1ul> jxl::N_AVX3_SPR::MaxWorkaround<hwy::N_AVX3_SPR::Vec128<float, 1ul> >(hwy::N_AVX3_SPR::Vec128<float, 1ul>, hwy::N_AVX3_SPR::Vec128<float, 1ul>)
Unexecuted instantiation: hwy::N_AVX3_SPR::Vec512<float> jxl::N_AVX3_SPR::MaxWorkaround<hwy::N_AVX3_SPR::Vec512<float> >(hwy::N_AVX3_SPR::Vec512<float>, hwy::N_AVX3_SPR::Vec512<float>)
Unexecuted instantiation: hwy::N_SSE2::Vec128<float, 1ul> jxl::N_SSE2::MaxWorkaround<hwy::N_SSE2::Vec128<float, 1ul> >(hwy::N_SSE2::Vec128<float, 1ul>, hwy::N_SSE2::Vec128<float, 1ul>)
Unexecuted instantiation: hwy::N_SSE2::Vec128<float, 4ul> jxl::N_SSE2::MaxWorkaround<hwy::N_SSE2::Vec128<float, 4ul> >(hwy::N_SSE2::Vec128<float, 4ul>, hwy::N_SSE2::Vec128<float, 4ul>)
63
64
template <typename D>
65
JXL_INLINE void ComputePixelChannel(const D d, const float dc_factor,
66
                                    const float* JXL_RESTRICT row_top,
67
                                    const float* JXL_RESTRICT row,
68
                                    const float* JXL_RESTRICT row_bottom,
69
                                    Vec<D>* JXL_RESTRICT mc,
70
                                    Vec<D>* JXL_RESTRICT sm,
71
12.0M
                                    Vec<D>* JXL_RESTRICT gap, size_t x) {
72
12.0M
  const auto tl = LoadU(d, row_top + x - 1);
73
12.0M
  const auto tc = Load(d, row_top + x);
74
12.0M
  const auto tr = LoadU(d, row_top + x + 1);
75
76
12.0M
  const auto ml = LoadU(d, row + x - 1);
77
12.0M
  *mc = Load(d, row + x);
78
12.0M
  const auto mr = LoadU(d, row + x + 1);
79
80
12.0M
  const auto bl = LoadU(d, row_bottom + x - 1);
81
12.0M
  const auto bc = Load(d, row_bottom + x);
82
12.0M
  const auto br = LoadU(d, row_bottom + x + 1);
83
84
12.0M
  const auto w_center = Set(d, w0);
85
12.0M
  const auto w_side = Set(d, w1);
86
12.0M
  const auto w_corner = Set(d, w2);
87
88
12.0M
  const auto corner = Add(Add(tl, tr), Add(bl, br));
89
12.0M
  const auto side = Add(Add(ml, mr), Add(tc, bc));
90
12.0M
  *sm = MulAdd(corner, w_corner, MulAdd(side, w_side, Mul(*mc, w_center)));
91
92
12.0M
  const auto dc_quant = Set(d, dc_factor);
93
12.0M
  *gap = MaxWorkaround(*gap, Abs(Div(Sub(*mc, *sm), dc_quant)));
94
12.0M
}
Unexecuted instantiation: void jxl::N_SSE4::ComputePixelChannel<hwy::N_SSE4::Simd<float, 1ul, 0> >(hwy::N_SSE4::Simd<float, 1ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_SSE4::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_SSE4::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_SSE4::Simd<float, 1ul, 0>)()))*, unsigned long)
Unexecuted instantiation: void jxl::N_SSE4::ComputePixelChannel<hwy::N_SSE4::Simd<float, 4ul, 0> >(hwy::N_SSE4::Simd<float, 4ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_SSE4::Simd<float, 4ul, 0>)()))*, decltype (Zero((hwy::N_SSE4::Simd<float, 4ul, 0>)()))*, decltype (Zero((hwy::N_SSE4::Simd<float, 4ul, 0>)()))*, unsigned long)
void jxl::N_AVX2::ComputePixelChannel<hwy::N_AVX2::Simd<float, 1ul, 0> >(hwy::N_AVX2::Simd<float, 1ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_AVX2::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_AVX2::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_AVX2::Simd<float, 1ul, 0>)()))*, unsigned long)
Line
Count
Source
71
9.70M
                                    Vec<D>* JXL_RESTRICT gap, size_t x) {
72
9.70M
  const auto tl = LoadU(d, row_top + x - 1);
73
9.70M
  const auto tc = Load(d, row_top + x);
74
9.70M
  const auto tr = LoadU(d, row_top + x + 1);
75
76
9.70M
  const auto ml = LoadU(d, row + x - 1);
77
9.70M
  *mc = Load(d, row + x);
78
9.70M
  const auto mr = LoadU(d, row + x + 1);
79
80
9.70M
  const auto bl = LoadU(d, row_bottom + x - 1);
81
9.70M
  const auto bc = Load(d, row_bottom + x);
82
9.70M
  const auto br = LoadU(d, row_bottom + x + 1);
83
84
9.70M
  const auto w_center = Set(d, w0);
85
9.70M
  const auto w_side = Set(d, w1);
86
9.70M
  const auto w_corner = Set(d, w2);
87
88
9.70M
  const auto corner = Add(Add(tl, tr), Add(bl, br));
89
9.70M
  const auto side = Add(Add(ml, mr), Add(tc, bc));
90
9.70M
  *sm = MulAdd(corner, w_corner, MulAdd(side, w_side, Mul(*mc, w_center)));
91
92
9.70M
  const auto dc_quant = Set(d, dc_factor);
93
9.70M
  *gap = MaxWorkaround(*gap, Abs(Div(Sub(*mc, *sm), dc_quant)));
94
9.70M
}
void jxl::N_AVX2::ComputePixelChannel<hwy::N_AVX2::Simd<float, 8ul, 0> >(hwy::N_AVX2::Simd<float, 8ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_AVX2::Simd<float, 8ul, 0>)()))*, decltype (Zero((hwy::N_AVX2::Simd<float, 8ul, 0>)()))*, decltype (Zero((hwy::N_AVX2::Simd<float, 8ul, 0>)()))*, unsigned long)
Line
Count
Source
71
2.36M
                                    Vec<D>* JXL_RESTRICT gap, size_t x) {
72
2.36M
  const auto tl = LoadU(d, row_top + x - 1);
73
2.36M
  const auto tc = Load(d, row_top + x);
74
2.36M
  const auto tr = LoadU(d, row_top + x + 1);
75
76
2.36M
  const auto ml = LoadU(d, row + x - 1);
77
2.36M
  *mc = Load(d, row + x);
78
2.36M
  const auto mr = LoadU(d, row + x + 1);
79
80
2.36M
  const auto bl = LoadU(d, row_bottom + x - 1);
81
2.36M
  const auto bc = Load(d, row_bottom + x);
82
2.36M
  const auto br = LoadU(d, row_bottom + x + 1);
83
84
2.36M
  const auto w_center = Set(d, w0);
85
2.36M
  const auto w_side = Set(d, w1);
86
2.36M
  const auto w_corner = Set(d, w2);
87
88
2.36M
  const auto corner = Add(Add(tl, tr), Add(bl, br));
89
2.36M
  const auto side = Add(Add(ml, mr), Add(tc, bc));
90
2.36M
  *sm = MulAdd(corner, w_corner, MulAdd(side, w_side, Mul(*mc, w_center)));
91
92
2.36M
  const auto dc_quant = Set(d, dc_factor);
93
2.36M
  *gap = MaxWorkaround(*gap, Abs(Div(Sub(*mc, *sm), dc_quant)));
94
2.36M
}
Unexecuted instantiation: void jxl::N_AVX3::ComputePixelChannel<hwy::N_AVX3::Simd<float, 1ul, 0> >(hwy::N_AVX3::Simd<float, 1ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_AVX3::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_AVX3::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_AVX3::Simd<float, 1ul, 0>)()))*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3::ComputePixelChannel<hwy::N_AVX3::Simd<float, 16ul, 0> >(hwy::N_AVX3::Simd<float, 16ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_AVX3::Simd<float, 16ul, 0>)()))*, decltype (Zero((hwy::N_AVX3::Simd<float, 16ul, 0>)()))*, decltype (Zero((hwy::N_AVX3::Simd<float, 16ul, 0>)()))*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3_ZEN4::ComputePixelChannel<hwy::N_AVX3_ZEN4::Simd<float, 1ul, 0> >(hwy::N_AVX3_ZEN4::Simd<float, 1ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_AVX3_ZEN4::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_AVX3_ZEN4::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_AVX3_ZEN4::Simd<float, 1ul, 0>)()))*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3_ZEN4::ComputePixelChannel<hwy::N_AVX3_ZEN4::Simd<float, 16ul, 0> >(hwy::N_AVX3_ZEN4::Simd<float, 16ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_AVX3_ZEN4::Simd<float, 16ul, 0>)()))*, decltype (Zero((hwy::N_AVX3_ZEN4::Simd<float, 16ul, 0>)()))*, decltype (Zero((hwy::N_AVX3_ZEN4::Simd<float, 16ul, 0>)()))*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3_SPR::ComputePixelChannel<hwy::N_AVX3_SPR::Simd<float, 1ul, 0> >(hwy::N_AVX3_SPR::Simd<float, 1ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_AVX3_SPR::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_AVX3_SPR::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_AVX3_SPR::Simd<float, 1ul, 0>)()))*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3_SPR::ComputePixelChannel<hwy::N_AVX3_SPR::Simd<float, 16ul, 0> >(hwy::N_AVX3_SPR::Simd<float, 16ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_AVX3_SPR::Simd<float, 16ul, 0>)()))*, decltype (Zero((hwy::N_AVX3_SPR::Simd<float, 16ul, 0>)()))*, decltype (Zero((hwy::N_AVX3_SPR::Simd<float, 16ul, 0>)()))*, unsigned long)
Unexecuted instantiation: void jxl::N_SSE2::ComputePixelChannel<hwy::N_SSE2::Simd<float, 1ul, 0> >(hwy::N_SSE2::Simd<float, 1ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_SSE2::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_SSE2::Simd<float, 1ul, 0>)()))*, decltype (Zero((hwy::N_SSE2::Simd<float, 1ul, 0>)()))*, unsigned long)
Unexecuted instantiation: void jxl::N_SSE2::ComputePixelChannel<hwy::N_SSE2::Simd<float, 4ul, 0> >(hwy::N_SSE2::Simd<float, 4ul, 0>, float, float const*, float const*, float const*, decltype (Zero((hwy::N_SSE2::Simd<float, 4ul, 0>)()))*, decltype (Zero((hwy::N_SSE2::Simd<float, 4ul, 0>)()))*, decltype (Zero((hwy::N_SSE2::Simd<float, 4ul, 0>)()))*, unsigned long)
95
96
template <typename D>
97
JXL_INLINE void ComputePixel(
98
    const float* JXL_RESTRICT dc_factors,
99
    const float* JXL_RESTRICT* JXL_RESTRICT rows_top,
100
    const float* JXL_RESTRICT* JXL_RESTRICT rows,
101
    const float* JXL_RESTRICT* JXL_RESTRICT rows_bottom,
102
4.02M
    float* JXL_RESTRICT* JXL_RESTRICT out_rows, size_t x) {
103
4.02M
  const D d;
104
4.02M
  auto mc_x = Undefined(d);
105
4.02M
  auto mc_y = Undefined(d);
106
4.02M
  auto mc_b = Undefined(d);
107
4.02M
  auto sm_x = Undefined(d);
108
4.02M
  auto sm_y = Undefined(d);
109
4.02M
  auto sm_b = Undefined(d);
110
4.02M
  auto gap = Set(d, 0.5f);
111
4.02M
  ComputePixelChannel(d, dc_factors[0], rows_top[0], rows[0], rows_bottom[0],
112
4.02M
                      &mc_x, &sm_x, &gap, x);
113
4.02M
  ComputePixelChannel(d, dc_factors[1], rows_top[1], rows[1], rows_bottom[1],
114
4.02M
                      &mc_y, &sm_y, &gap, x);
115
4.02M
  ComputePixelChannel(d, dc_factors[2], rows_top[2], rows[2], rows_bottom[2],
116
4.02M
                      &mc_b, &sm_b, &gap, x);
117
4.02M
  auto factor = MulAdd(Set(d, -4.0f), gap, Set(d, 3.0f));
118
4.02M
  factor = ZeroIfNegative(factor);
119
120
4.02M
  auto out = MulAdd(Sub(sm_x, mc_x), factor, mc_x);
121
4.02M
  Store(out, d, out_rows[0] + x);
122
4.02M
  out = MulAdd(Sub(sm_y, mc_y), factor, mc_y);
123
4.02M
  Store(out, d, out_rows[1] + x);
124
4.02M
  out = MulAdd(Sub(sm_b, mc_b), factor, mc_b);
125
4.02M
  Store(out, d, out_rows[2] + x);
126
4.02M
}
Unexecuted instantiation: void jxl::N_SSE4::ComputePixel<hwy::N_SSE4::Simd<float, 1ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Unexecuted instantiation: void jxl::N_SSE4::ComputePixel<hwy::N_SSE4::Simd<float, 4ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
void jxl::N_AVX2::ComputePixel<hwy::N_AVX2::Simd<float, 1ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Line
Count
Source
102
3.23M
    float* JXL_RESTRICT* JXL_RESTRICT out_rows, size_t x) {
103
3.23M
  const D d;
104
3.23M
  auto mc_x = Undefined(d);
105
3.23M
  auto mc_y = Undefined(d);
106
3.23M
  auto mc_b = Undefined(d);
107
3.23M
  auto sm_x = Undefined(d);
108
3.23M
  auto sm_y = Undefined(d);
109
3.23M
  auto sm_b = Undefined(d);
110
3.23M
  auto gap = Set(d, 0.5f);
111
3.23M
  ComputePixelChannel(d, dc_factors[0], rows_top[0], rows[0], rows_bottom[0],
112
3.23M
                      &mc_x, &sm_x, &gap, x);
113
3.23M
  ComputePixelChannel(d, dc_factors[1], rows_top[1], rows[1], rows_bottom[1],
114
3.23M
                      &mc_y, &sm_y, &gap, x);
115
3.23M
  ComputePixelChannel(d, dc_factors[2], rows_top[2], rows[2], rows_bottom[2],
116
3.23M
                      &mc_b, &sm_b, &gap, x);
117
3.23M
  auto factor = MulAdd(Set(d, -4.0f), gap, Set(d, 3.0f));
118
3.23M
  factor = ZeroIfNegative(factor);
119
120
3.23M
  auto out = MulAdd(Sub(sm_x, mc_x), factor, mc_x);
121
3.23M
  Store(out, d, out_rows[0] + x);
122
3.23M
  out = MulAdd(Sub(sm_y, mc_y), factor, mc_y);
123
3.23M
  Store(out, d, out_rows[1] + x);
124
3.23M
  out = MulAdd(Sub(sm_b, mc_b), factor, mc_b);
125
3.23M
  Store(out, d, out_rows[2] + x);
126
3.23M
}
void jxl::N_AVX2::ComputePixel<hwy::N_AVX2::Simd<float, 8ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Line
Count
Source
102
786k
    float* JXL_RESTRICT* JXL_RESTRICT out_rows, size_t x) {
103
786k
  const D d;
104
786k
  auto mc_x = Undefined(d);
105
786k
  auto mc_y = Undefined(d);
106
786k
  auto mc_b = Undefined(d);
107
786k
  auto sm_x = Undefined(d);
108
786k
  auto sm_y = Undefined(d);
109
786k
  auto sm_b = Undefined(d);
110
786k
  auto gap = Set(d, 0.5f);
111
786k
  ComputePixelChannel(d, dc_factors[0], rows_top[0], rows[0], rows_bottom[0],
112
786k
                      &mc_x, &sm_x, &gap, x);
113
786k
  ComputePixelChannel(d, dc_factors[1], rows_top[1], rows[1], rows_bottom[1],
114
786k
                      &mc_y, &sm_y, &gap, x);
115
786k
  ComputePixelChannel(d, dc_factors[2], rows_top[2], rows[2], rows_bottom[2],
116
786k
                      &mc_b, &sm_b, &gap, x);
117
786k
  auto factor = MulAdd(Set(d, -4.0f), gap, Set(d, 3.0f));
118
786k
  factor = ZeroIfNegative(factor);
119
120
786k
  auto out = MulAdd(Sub(sm_x, mc_x), factor, mc_x);
121
786k
  Store(out, d, out_rows[0] + x);
122
786k
  out = MulAdd(Sub(sm_y, mc_y), factor, mc_y);
123
786k
  Store(out, d, out_rows[1] + x);
124
786k
  out = MulAdd(Sub(sm_b, mc_b), factor, mc_b);
125
786k
  Store(out, d, out_rows[2] + x);
126
786k
}
Unexecuted instantiation: void jxl::N_AVX3::ComputePixel<hwy::N_AVX3::Simd<float, 1ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3::ComputePixel<hwy::N_AVX3::Simd<float, 16ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3_ZEN4::ComputePixel<hwy::N_AVX3_ZEN4::Simd<float, 1ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3_ZEN4::ComputePixel<hwy::N_AVX3_ZEN4::Simd<float, 16ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3_SPR::ComputePixel<hwy::N_AVX3_SPR::Simd<float, 1ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Unexecuted instantiation: void jxl::N_AVX3_SPR::ComputePixel<hwy::N_AVX3_SPR::Simd<float, 16ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Unexecuted instantiation: void jxl::N_SSE2::ComputePixel<hwy::N_SSE2::Simd<float, 1ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
Unexecuted instantiation: void jxl::N_SSE2::ComputePixel<hwy::N_SSE2::Simd<float, 4ul, 0> >(float const*, float const* restrict*, float const* restrict*, float const* restrict*, float* restrict*, unsigned long)
127
128
Status AdaptiveDCSmoothing(JxlMemoryManager* memory_manager,
129
                           const float* dc_factors, Image3F* dc,
130
44.1k
                           ThreadPool* pool) {
131
44.1k
  const size_t xsize = dc->xsize();
132
44.1k
  const size_t ysize = dc->ysize();
133
44.1k
  if (ysize <= 2 || xsize <= 2) return true;
134
135
  // TODO(veluca): use tile-based processing?
136
  // TODO(veluca): decide if changes to the y channel should be propagated to
137
  // the x and b channels through color correlation.
138
7.55k
  JXL_ENSURE(w1 + w2 < 0.25f);
139
140
15.1k
  JXL_ASSIGN_OR_RETURN(Image3F smoothed,
141
15.1k
                       Image3F::Create(memory_manager, xsize, ysize));
142
  // Fill in borders that the loop below will not. First and last are unused.
143
30.2k
  for (size_t c = 0; c < 3; c++) {
144
45.3k
    for (size_t y : {static_cast<size_t>(0), ysize - 1}) {
145
45.3k
      memcpy(smoothed.PlaneRow(c, y), dc->PlaneRow(c, y),
146
45.3k
             xsize * sizeof(float));
147
45.3k
    }
148
22.6k
  }
149
306k
  auto process_row = [&](const uint32_t y, size_t /*thread*/) -> Status {
150
306k
    const float* JXL_RESTRICT rows_top[3]{
151
306k
        dc->ConstPlaneRow(0, y - 1),
152
306k
        dc->ConstPlaneRow(1, y - 1),
153
306k
        dc->ConstPlaneRow(2, y - 1),
154
306k
    };
155
306k
    const float* JXL_RESTRICT rows[3] = {
156
306k
        dc->ConstPlaneRow(0, y),
157
306k
        dc->ConstPlaneRow(1, y),
158
306k
        dc->ConstPlaneRow(2, y),
159
306k
    };
160
306k
    const float* JXL_RESTRICT rows_bottom[3] = {
161
306k
        dc->ConstPlaneRow(0, y + 1),
162
306k
        dc->ConstPlaneRow(1, y + 1),
163
306k
        dc->ConstPlaneRow(2, y + 1),
164
306k
    };
165
306k
    float* JXL_RESTRICT rows_out[3] = {
166
306k
        smoothed.PlaneRow(0, y),
167
306k
        smoothed.PlaneRow(1, y),
168
306k
        smoothed.PlaneRow(2, y),
169
306k
    };
170
612k
    for (size_t x : {static_cast<size_t>(0), xsize - 1}) {
171
2.44M
      for (size_t c = 0; c < 3; c++) {
172
1.83M
        rows_out[c][x] = rows[c][x];
173
1.83M
      }
174
612k
    }
175
176
306k
    size_t x = 1;
177
    // First pixels
178
306k
    const size_t N = Lanes(D());
179
2.42M
    for (; x < std::min(N, xsize - 1); x++) {
180
2.11M
      ComputePixel<DScalar>(dc_factors, rows_top, rows, rows_bottom, rows_out,
181
2.11M
                            x);
182
2.11M
    }
183
    // Full vectors.
184
1.09M
    for (; x + N <= xsize - 1; x += N) {
185
786k
      ComputePixel<D>(dc_factors, rows_top, rows, rows_bottom, rows_out, x);
186
786k
    }
187
    // Last pixels.
188
1.42M
    for (; x < xsize - 1; x++) {
189
1.11M
      ComputePixel<DScalar>(dc_factors, rows_top, rows, rows_bottom, rows_out,
190
1.11M
                            x);
191
1.11M
    }
192
306k
    return true;
193
306k
  };
Unexecuted instantiation: compressed_dc.cc:jxl::N_SSE4::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)::$_0::operator()(unsigned int, unsigned long) const
compressed_dc.cc:jxl::N_AVX2::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)::$_0::operator()(unsigned int, unsigned long) const
Line
Count
Source
149
306k
  auto process_row = [&](const uint32_t y, size_t /*thread*/) -> Status {
150
306k
    const float* JXL_RESTRICT rows_top[3]{
151
306k
        dc->ConstPlaneRow(0, y - 1),
152
306k
        dc->ConstPlaneRow(1, y - 1),
153
306k
        dc->ConstPlaneRow(2, y - 1),
154
306k
    };
155
306k
    const float* JXL_RESTRICT rows[3] = {
156
306k
        dc->ConstPlaneRow(0, y),
157
306k
        dc->ConstPlaneRow(1, y),
158
306k
        dc->ConstPlaneRow(2, y),
159
306k
    };
160
306k
    const float* JXL_RESTRICT rows_bottom[3] = {
161
306k
        dc->ConstPlaneRow(0, y + 1),
162
306k
        dc->ConstPlaneRow(1, y + 1),
163
306k
        dc->ConstPlaneRow(2, y + 1),
164
306k
    };
165
306k
    float* JXL_RESTRICT rows_out[3] = {
166
306k
        smoothed.PlaneRow(0, y),
167
306k
        smoothed.PlaneRow(1, y),
168
306k
        smoothed.PlaneRow(2, y),
169
306k
    };
170
612k
    for (size_t x : {static_cast<size_t>(0), xsize - 1}) {
171
2.44M
      for (size_t c = 0; c < 3; c++) {
172
1.83M
        rows_out[c][x] = rows[c][x];
173
1.83M
      }
174
612k
    }
175
176
306k
    size_t x = 1;
177
    // First pixels
178
306k
    const size_t N = Lanes(D());
179
2.42M
    for (; x < std::min(N, xsize - 1); x++) {
180
2.11M
      ComputePixel<DScalar>(dc_factors, rows_top, rows, rows_bottom, rows_out,
181
2.11M
                            x);
182
2.11M
    }
183
    // Full vectors.
184
1.09M
    for (; x + N <= xsize - 1; x += N) {
185
786k
      ComputePixel<D>(dc_factors, rows_top, rows, rows_bottom, rows_out, x);
186
786k
    }
187
    // Last pixels.
188
1.42M
    for (; x < xsize - 1; x++) {
189
1.11M
      ComputePixel<DScalar>(dc_factors, rows_top, rows, rows_bottom, rows_out,
190
1.11M
                            x);
191
1.11M
    }
192
306k
    return true;
193
306k
  };
Unexecuted instantiation: compressed_dc.cc:jxl::N_AVX3::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)::$_0::operator()(unsigned int, unsigned long) const
Unexecuted instantiation: compressed_dc.cc:jxl::N_AVX3_ZEN4::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)::$_0::operator()(unsigned int, unsigned long) const
Unexecuted instantiation: compressed_dc.cc:jxl::N_AVX3_SPR::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)::$_0::operator()(unsigned int, unsigned long) const
Unexecuted instantiation: compressed_dc.cc:jxl::N_SSE2::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)::$_0::operator()(unsigned int, unsigned long) const
194
15.1k
  JXL_RETURN_IF_ERROR(RunOnPool(pool, 1, ysize - 1, ThreadPool::NoInit,
195
15.1k
                                process_row, "DCSmoothingRow"));
196
7.55k
  dc->Swap(smoothed);
197
7.55k
  return true;
198
15.1k
}
Unexecuted instantiation: jxl::N_SSE4::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)
jxl::N_AVX2::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)
Line
Count
Source
130
44.1k
                           ThreadPool* pool) {
131
44.1k
  const size_t xsize = dc->xsize();
132
44.1k
  const size_t ysize = dc->ysize();
133
44.1k
  if (ysize <= 2 || xsize <= 2) return true;
134
135
  // TODO(veluca): use tile-based processing?
136
  // TODO(veluca): decide if changes to the y channel should be propagated to
137
  // the x and b channels through color correlation.
138
7.55k
  JXL_ENSURE(w1 + w2 < 0.25f);
139
140
15.1k
  JXL_ASSIGN_OR_RETURN(Image3F smoothed,
141
15.1k
                       Image3F::Create(memory_manager, xsize, ysize));
142
  // Fill in borders that the loop below will not. First and last are unused.
143
30.2k
  for (size_t c = 0; c < 3; c++) {
144
45.3k
    for (size_t y : {static_cast<size_t>(0), ysize - 1}) {
145
45.3k
      memcpy(smoothed.PlaneRow(c, y), dc->PlaneRow(c, y),
146
45.3k
             xsize * sizeof(float));
147
45.3k
    }
148
22.6k
  }
149
15.1k
  auto process_row = [&](const uint32_t y, size_t /*thread*/) -> Status {
150
15.1k
    const float* JXL_RESTRICT rows_top[3]{
151
15.1k
        dc->ConstPlaneRow(0, y - 1),
152
15.1k
        dc->ConstPlaneRow(1, y - 1),
153
15.1k
        dc->ConstPlaneRow(2, y - 1),
154
15.1k
    };
155
15.1k
    const float* JXL_RESTRICT rows[3] = {
156
15.1k
        dc->ConstPlaneRow(0, y),
157
15.1k
        dc->ConstPlaneRow(1, y),
158
15.1k
        dc->ConstPlaneRow(2, y),
159
15.1k
    };
160
15.1k
    const float* JXL_RESTRICT rows_bottom[3] = {
161
15.1k
        dc->ConstPlaneRow(0, y + 1),
162
15.1k
        dc->ConstPlaneRow(1, y + 1),
163
15.1k
        dc->ConstPlaneRow(2, y + 1),
164
15.1k
    };
165
15.1k
    float* JXL_RESTRICT rows_out[3] = {
166
15.1k
        smoothed.PlaneRow(0, y),
167
15.1k
        smoothed.PlaneRow(1, y),
168
15.1k
        smoothed.PlaneRow(2, y),
169
15.1k
    };
170
15.1k
    for (size_t x : {static_cast<size_t>(0), xsize - 1}) {
171
15.1k
      for (size_t c = 0; c < 3; c++) {
172
15.1k
        rows_out[c][x] = rows[c][x];
173
15.1k
      }
174
15.1k
    }
175
176
15.1k
    size_t x = 1;
177
    // First pixels
178
15.1k
    const size_t N = Lanes(D());
179
15.1k
    for (; x < std::min(N, xsize - 1); x++) {
180
15.1k
      ComputePixel<DScalar>(dc_factors, rows_top, rows, rows_bottom, rows_out,
181
15.1k
                            x);
182
15.1k
    }
183
    // Full vectors.
184
15.1k
    for (; x + N <= xsize - 1; x += N) {
185
15.1k
      ComputePixel<D>(dc_factors, rows_top, rows, rows_bottom, rows_out, x);
186
15.1k
    }
187
    // Last pixels.
188
15.1k
    for (; x < xsize - 1; x++) {
189
15.1k
      ComputePixel<DScalar>(dc_factors, rows_top, rows, rows_bottom, rows_out,
190
15.1k
                            x);
191
15.1k
    }
192
15.1k
    return true;
193
15.1k
  };
194
15.1k
  JXL_RETURN_IF_ERROR(RunOnPool(pool, 1, ysize - 1, ThreadPool::NoInit,
195
15.1k
                                process_row, "DCSmoothingRow"));
196
7.55k
  dc->Swap(smoothed);
197
7.55k
  return true;
198
15.1k
}
Unexecuted instantiation: jxl::N_AVX3::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)
Unexecuted instantiation: jxl::N_AVX3_ZEN4::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)
Unexecuted instantiation: jxl::N_AVX3_SPR::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)
Unexecuted instantiation: jxl::N_SSE2::AdaptiveDCSmoothing(JxlMemoryManagerStruct*, float const*, jxl::Image3<float>*, jxl::ThreadPool*)
199
200
// DC dequantization.
201
void DequantDC(const Rect& r, Image3F* dc, ImageB* quant_dc, const Image& in,
202
               const float* dc_factors, float mul, const float* cfl_factors,
203
               const YCbCrChromaSubsampling& chroma_subsampling,
204
67.6k
               const BlockCtxMap& bctx) {
205
67.6k
  const HWY_FULL(float) df;
206
67.6k
  const Rebind<pixel_type, HWY_FULL(float)> di;  // assumes pixel_type <= float
207
67.6k
  if (chroma_subsampling.Is444()) {
208
62.3k
    const auto fac_x = Set(df, dc_factors[0] * mul);
209
62.3k
    const auto fac_y = Set(df, dc_factors[1] * mul);
210
62.3k
    const auto fac_b = Set(df, dc_factors[2] * mul);
211
62.3k
    const auto cfl_fac_x = Set(df, cfl_factors[0]);
212
62.3k
    const auto cfl_fac_b = Set(df, cfl_factors[2]);
213
608k
    for (size_t y = 0; y < r.ysize(); y++) {
214
545k
      float* dec_row_x = r.PlaneRow(dc, 0, y);
215
545k
      float* dec_row_y = r.PlaneRow(dc, 1, y);
216
545k
      float* dec_row_b = r.PlaneRow(dc, 2, y);
217
545k
      const int32_t* quant_row_x = in.channel[1].plane.Row(y);
218
545k
      const int32_t* quant_row_y = in.channel[0].plane.Row(y);
219
545k
      const int32_t* quant_row_b = in.channel[2].plane.Row(y);
220
2.30M
      for (size_t x = 0; x < r.xsize(); x += Lanes(di)) {
221
1.75M
        const auto in_q_x = Load(di, quant_row_x + x);
222
1.75M
        const auto in_q_y = Load(di, quant_row_y + x);
223
1.75M
        const auto in_q_b = Load(di, quant_row_b + x);
224
1.75M
        const auto in_x = Mul(ConvertTo(df, in_q_x), fac_x);
225
1.75M
        const auto in_y = Mul(ConvertTo(df, in_q_y), fac_y);
226
1.75M
        const auto in_b = Mul(ConvertTo(df, in_q_b), fac_b);
227
1.75M
        Store(in_y, df, dec_row_y + x);
228
1.75M
        Store(MulAdd(in_y, cfl_fac_x, in_x), df, dec_row_x + x);
229
1.75M
        Store(MulAdd(in_y, cfl_fac_b, in_b), df, dec_row_b + x);
230
1.75M
      }
231
545k
    }
232
62.3k
  } else {
233
16.0k
    for (size_t c : {1, 0, 2}) {
234
16.0k
      Rect rect(r.x0() >> chroma_subsampling.HShift(c),
235
16.0k
                r.y0() >> chroma_subsampling.VShift(c),
236
16.0k
                r.xsize() >> chroma_subsampling.HShift(c),
237
16.0k
                r.ysize() >> chroma_subsampling.VShift(c));
238
16.0k
      const auto fac = Set(df, dc_factors[c] * mul);
239
16.0k
      const Channel& ch = in.channel[c < 2 ? c ^ 1 : c];
240
546k
      for (size_t y = 0; y < rect.ysize(); y++) {
241
530k
        const int32_t* quant_row = ch.plane.Row(y);
242
530k
        float* row = rect.PlaneRow(dc, c, y);
243
1.22M
        for (size_t x = 0; x < rect.xsize(); x += Lanes(di)) {
244
698k
          const auto in_q = Load(di, quant_row + x);
245
698k
          const auto out = Mul(ConvertTo(df, in_q), fac);
246
698k
          Store(out, df, row + x);
247
698k
        }
248
530k
      }
249
16.0k
    }
250
5.33k
  }
251
67.6k
  if (bctx.num_dc_ctxs <= 1) {
252
636k
    for (size_t y = 0; y < r.ysize(); y++) {
253
574k
      uint8_t* qdc_row = r.Row(quant_dc, y);
254
574k
      memset(qdc_row, 0, sizeof(*qdc_row) * r.xsize());
255
574k
    }
256
62.0k
  } else {
257
5.59k
    JXL_DASSERT(r.ysize() == 0 ||
258
5.59k
                (r.ysize() - 1) >> chroma_subsampling.VShift(0) <
259
5.59k
                    in.channel[1].plane.ysize());
260
5.59k
    JXL_DASSERT(r.ysize() == 0 ||
261
5.59k
                (r.ysize() - 1) >> chroma_subsampling.VShift(1) <
262
5.59k
                    in.channel[0].plane.ysize());
263
5.59k
    JXL_DASSERT(r.ysize() == 0 ||
264
5.59k
                (r.ysize() - 1) >> chroma_subsampling.VShift(2) <
265
5.59k
                    in.channel[2].plane.ysize());
266
154k
    for (size_t y = 0; y < r.ysize(); y++) {
267
148k
      uint8_t* qdc_row_val = r.Row(quant_dc, y);
268
148k
      const int32_t* quant_row_x =
269
148k
          in.channel[1].plane.Row(y >> chroma_subsampling.VShift(0));
270
148k
      const int32_t* quant_row_y =
271
148k
          in.channel[0].plane.Row(y >> chroma_subsampling.VShift(1));
272
148k
      const int32_t* quant_row_b =
273
148k
          in.channel[2].plane.Row(y >> chroma_subsampling.VShift(2));
274
1.63M
      for (size_t x = 0; x < r.xsize(); x++) {
275
1.48M
        int bucket_x = 0;
276
1.48M
        int bucket_y = 0;
277
1.48M
        int bucket_b = 0;
278
8.45M
        for (int t : bctx.dc_thresholds[0]) {
279
8.45M
          if (quant_row_x[x >> chroma_subsampling.HShift(0)] > t) bucket_x++;
280
8.45M
        }
281
1.48M
        for (int t : bctx.dc_thresholds[1]) {
282
208k
          if (quant_row_y[x >> chroma_subsampling.HShift(1)] > t) bucket_y++;
283
208k
        }
284
1.48M
        for (int t : bctx.dc_thresholds[2]) {
285
183k
          if (quant_row_b[x >> chroma_subsampling.HShift(2)] > t) bucket_b++;
286
183k
        }
287
1.48M
        int bucket = bucket_x;
288
1.48M
        bucket *= bctx.dc_thresholds[2].size() + 1;
289
1.48M
        bucket += bucket_b;
290
1.48M
        bucket *= bctx.dc_thresholds[1].size() + 1;
291
1.48M
        bucket += bucket_y;
292
1.48M
        qdc_row_val[x] = bucket;
293
1.48M
      }
294
148k
    }
295
5.59k
  }
296
67.6k
}
Unexecuted instantiation: jxl::N_SSE4::DequantDC(jxl::RectT<unsigned long> const&, jxl::Image3<float>*, jxl::Plane<unsigned char>*, jxl::Image const&, float const*, float, float const*, jxl::YCbCrChromaSubsampling const&, jxl::BlockCtxMap const&)
jxl::N_AVX2::DequantDC(jxl::RectT<unsigned long> const&, jxl::Image3<float>*, jxl::Plane<unsigned char>*, jxl::Image const&, float const*, float, float const*, jxl::YCbCrChromaSubsampling const&, jxl::BlockCtxMap const&)
Line
Count
Source
204
67.6k
               const BlockCtxMap& bctx) {
205
67.6k
  const HWY_FULL(float) df;
206
67.6k
  const Rebind<pixel_type, HWY_FULL(float)> di;  // assumes pixel_type <= float
207
67.6k
  if (chroma_subsampling.Is444()) {
208
62.3k
    const auto fac_x = Set(df, dc_factors[0] * mul);
209
62.3k
    const auto fac_y = Set(df, dc_factors[1] * mul);
210
62.3k
    const auto fac_b = Set(df, dc_factors[2] * mul);
211
62.3k
    const auto cfl_fac_x = Set(df, cfl_factors[0]);
212
62.3k
    const auto cfl_fac_b = Set(df, cfl_factors[2]);
213
608k
    for (size_t y = 0; y < r.ysize(); y++) {
214
545k
      float* dec_row_x = r.PlaneRow(dc, 0, y);
215
545k
      float* dec_row_y = r.PlaneRow(dc, 1, y);
216
545k
      float* dec_row_b = r.PlaneRow(dc, 2, y);
217
545k
      const int32_t* quant_row_x = in.channel[1].plane.Row(y);
218
545k
      const int32_t* quant_row_y = in.channel[0].plane.Row(y);
219
545k
      const int32_t* quant_row_b = in.channel[2].plane.Row(y);
220
2.30M
      for (size_t x = 0; x < r.xsize(); x += Lanes(di)) {
221
1.75M
        const auto in_q_x = Load(di, quant_row_x + x);
222
1.75M
        const auto in_q_y = Load(di, quant_row_y + x);
223
1.75M
        const auto in_q_b = Load(di, quant_row_b + x);
224
1.75M
        const auto in_x = Mul(ConvertTo(df, in_q_x), fac_x);
225
1.75M
        const auto in_y = Mul(ConvertTo(df, in_q_y), fac_y);
226
1.75M
        const auto in_b = Mul(ConvertTo(df, in_q_b), fac_b);
227
1.75M
        Store(in_y, df, dec_row_y + x);
228
1.75M
        Store(MulAdd(in_y, cfl_fac_x, in_x), df, dec_row_x + x);
229
1.75M
        Store(MulAdd(in_y, cfl_fac_b, in_b), df, dec_row_b + x);
230
1.75M
      }
231
545k
    }
232
62.3k
  } else {
233
16.0k
    for (size_t c : {1, 0, 2}) {
234
16.0k
      Rect rect(r.x0() >> chroma_subsampling.HShift(c),
235
16.0k
                r.y0() >> chroma_subsampling.VShift(c),
236
16.0k
                r.xsize() >> chroma_subsampling.HShift(c),
237
16.0k
                r.ysize() >> chroma_subsampling.VShift(c));
238
16.0k
      const auto fac = Set(df, dc_factors[c] * mul);
239
16.0k
      const Channel& ch = in.channel[c < 2 ? c ^ 1 : c];
240
546k
      for (size_t y = 0; y < rect.ysize(); y++) {
241
530k
        const int32_t* quant_row = ch.plane.Row(y);
242
530k
        float* row = rect.PlaneRow(dc, c, y);
243
1.22M
        for (size_t x = 0; x < rect.xsize(); x += Lanes(di)) {
244
698k
          const auto in_q = Load(di, quant_row + x);
245
698k
          const auto out = Mul(ConvertTo(df, in_q), fac);
246
698k
          Store(out, df, row + x);
247
698k
        }
248
530k
      }
249
16.0k
    }
250
5.33k
  }
251
67.6k
  if (bctx.num_dc_ctxs <= 1) {
252
636k
    for (size_t y = 0; y < r.ysize(); y++) {
253
574k
      uint8_t* qdc_row = r.Row(quant_dc, y);
254
574k
      memset(qdc_row, 0, sizeof(*qdc_row) * r.xsize());
255
574k
    }
256
62.0k
  } else {
257
5.59k
    JXL_DASSERT(r.ysize() == 0 ||
258
5.59k
                (r.ysize() - 1) >> chroma_subsampling.VShift(0) <
259
5.59k
                    in.channel[1].plane.ysize());
260
5.59k
    JXL_DASSERT(r.ysize() == 0 ||
261
5.59k
                (r.ysize() - 1) >> chroma_subsampling.VShift(1) <
262
5.59k
                    in.channel[0].plane.ysize());
263
5.59k
    JXL_DASSERT(r.ysize() == 0 ||
264
5.59k
                (r.ysize() - 1) >> chroma_subsampling.VShift(2) <
265
5.59k
                    in.channel[2].plane.ysize());
266
154k
    for (size_t y = 0; y < r.ysize(); y++) {
267
148k
      uint8_t* qdc_row_val = r.Row(quant_dc, y);
268
148k
      const int32_t* quant_row_x =
269
148k
          in.channel[1].plane.Row(y >> chroma_subsampling.VShift(0));
270
148k
      const int32_t* quant_row_y =
271
148k
          in.channel[0].plane.Row(y >> chroma_subsampling.VShift(1));
272
148k
      const int32_t* quant_row_b =
273
148k
          in.channel[2].plane.Row(y >> chroma_subsampling.VShift(2));
274
1.63M
      for (size_t x = 0; x < r.xsize(); x++) {
275
1.48M
        int bucket_x = 0;
276
1.48M
        int bucket_y = 0;
277
1.48M
        int bucket_b = 0;
278
8.45M
        for (int t : bctx.dc_thresholds[0]) {
279
8.45M
          if (quant_row_x[x >> chroma_subsampling.HShift(0)] > t) bucket_x++;
280
8.45M
        }
281
1.48M
        for (int t : bctx.dc_thresholds[1]) {
282
208k
          if (quant_row_y[x >> chroma_subsampling.HShift(1)] > t) bucket_y++;
283
208k
        }
284
1.48M
        for (int t : bctx.dc_thresholds[2]) {
285
183k
          if (quant_row_b[x >> chroma_subsampling.HShift(2)] > t) bucket_b++;
286
183k
        }
287
1.48M
        int bucket = bucket_x;
288
1.48M
        bucket *= bctx.dc_thresholds[2].size() + 1;
289
1.48M
        bucket += bucket_b;
290
1.48M
        bucket *= bctx.dc_thresholds[1].size() + 1;
291
1.48M
        bucket += bucket_y;
292
1.48M
        qdc_row_val[x] = bucket;
293
1.48M
      }
294
148k
    }
295
5.59k
  }
296
67.6k
}
Unexecuted instantiation: jxl::N_AVX3::DequantDC(jxl::RectT<unsigned long> const&, jxl::Image3<float>*, jxl::Plane<unsigned char>*, jxl::Image const&, float const*, float, float const*, jxl::YCbCrChromaSubsampling const&, jxl::BlockCtxMap const&)
Unexecuted instantiation: jxl::N_AVX3_ZEN4::DequantDC(jxl::RectT<unsigned long> const&, jxl::Image3<float>*, jxl::Plane<unsigned char>*, jxl::Image const&, float const*, float, float const*, jxl::YCbCrChromaSubsampling const&, jxl::BlockCtxMap const&)
Unexecuted instantiation: jxl::N_AVX3_SPR::DequantDC(jxl::RectT<unsigned long> const&, jxl::Image3<float>*, jxl::Plane<unsigned char>*, jxl::Image const&, float const*, float, float const*, jxl::YCbCrChromaSubsampling const&, jxl::BlockCtxMap const&)
Unexecuted instantiation: jxl::N_SSE2::DequantDC(jxl::RectT<unsigned long> const&, jxl::Image3<float>*, jxl::Plane<unsigned char>*, jxl::Image const&, float const*, float, float const*, jxl::YCbCrChromaSubsampling const&, jxl::BlockCtxMap const&)
297
298
// NOLINTNEXTLINE(google-readability-namespace-comments)
299
}  // namespace HWY_NAMESPACE
300
}  // namespace jxl
301
HWY_AFTER_NAMESPACE();
302
303
#if HWY_ONCE
304
namespace jxl {
305
306
HWY_EXPORT(DequantDC);
307
HWY_EXPORT(AdaptiveDCSmoothing);
308
Status AdaptiveDCSmoothing(JxlMemoryManager* memory_manager,
309
                           const float* dc_factors, Image3F* dc,
310
44.1k
                           ThreadPool* pool) {
311
44.1k
  return HWY_DYNAMIC_DISPATCH(AdaptiveDCSmoothing)(memory_manager, dc_factors,
312
44.1k
                                                   dc, pool);
313
44.1k
}
314
315
void DequantDC(const Rect& r, Image3F* dc, ImageB* quant_dc, const Image& in,
316
               const float* dc_factors, float mul, const float* cfl_factors,
317
               const YCbCrChromaSubsampling& chroma_subsampling,
318
67.6k
               const BlockCtxMap& bctx) {
319
67.6k
  HWY_DYNAMIC_DISPATCH(DequantDC)
320
67.6k
  (r, dc, quant_dc, in, dc_factors, mul, cfl_factors, chroma_subsampling, bctx);
321
67.6k
}
322
323
}  // namespace jxl
324
#endif  // HWY_ONCE