/src/libjxl/lib/jxl/modular/transform/rct.cc
Line | Count | Source |
1 | | // Copyright (c) the JPEG XL Project Authors. All rights reserved. |
2 | | // |
3 | | // Use of this source code is governed by a BSD-style |
4 | | // license that can be found in the LICENSE file. |
5 | | |
6 | | #include "lib/jxl/modular/transform/rct.h" |
7 | | |
8 | | #include <cstddef> |
9 | | #include <cstdint> |
10 | | #include <utility> |
11 | | |
12 | | #include "lib/jxl/base/data_parallel.h" |
13 | | #include "lib/jxl/base/status.h" |
14 | | #include "lib/jxl/modular/modular_image.h" |
15 | | #include "lib/jxl/modular/transform/transform.h" |
16 | | #undef HWY_TARGET_INCLUDE |
17 | | #define HWY_TARGET_INCLUDE "lib/jxl/modular/transform/rct.cc" |
18 | | #include <hwy/foreach_target.h> |
19 | | #include <hwy/highway.h> |
20 | | HWY_BEFORE_NAMESPACE(); |
21 | | namespace jxl { |
22 | | namespace HWY_NAMESPACE { |
23 | | |
24 | | // These templates are not found via ADL. |
25 | | using hwy::HWY_NAMESPACE::Add; |
26 | | using hwy::HWY_NAMESPACE::ShiftRight; |
27 | | using hwy::HWY_NAMESPACE::Sub; |
28 | | |
29 | | template <int transform_type> |
30 | | void InvRCTRow(const pixel_type* in0, const pixel_type* in1, |
31 | | const pixel_type* in2, pixel_type* out0, pixel_type* out1, |
32 | 85.9k | pixel_type* out2, size_t w) { |
33 | 85.9k | static_assert(transform_type >= 0 && transform_type < 7, |
34 | 85.9k | "Invalid transform type"); |
35 | 85.9k | int second = transform_type >> 1; |
36 | 85.9k | int third = transform_type & 1; |
37 | | |
38 | 85.9k | size_t x = 0; |
39 | 85.9k | const HWY_FULL(pixel_type) d; |
40 | 85.9k | const size_t N = Lanes(d); |
41 | 1.06M | for (; x + N - 1 < w; x += N) { |
42 | 974k | if (transform_type == 6) { |
43 | 595k | auto Y = Load(d, in0 + x); |
44 | 595k | auto Co = Load(d, in1 + x); |
45 | 595k | auto Cg = Load(d, in2 + x); |
46 | 595k | Y = Sub(Y, ShiftRight<1>(Cg)); |
47 | 595k | auto G = Add(Cg, Y); |
48 | 595k | Y = Sub(Y, ShiftRight<1>(Co)); |
49 | 595k | auto R = Add(Y, Co); |
50 | 595k | Store(R, d, out0 + x); |
51 | 595k | Store(G, d, out1 + x); |
52 | 595k | Store(Y, d, out2 + x); |
53 | 595k | } else { |
54 | 378k | auto First = Load(d, in0 + x); |
55 | 378k | auto Second = Load(d, in1 + x); |
56 | 378k | auto Third = Load(d, in2 + x); |
57 | 378k | if (third) Third = Add(Third, First); |
58 | 378k | if (second == 1) { |
59 | 333k | Second = Add(Second, First); |
60 | 333k | } else if (second == 2) { |
61 | 17.0k | Second = Add(Second, ShiftRight<1>(Add(First, Third))); |
62 | 17.0k | } |
63 | 378k | Store(First, d, out0 + x); |
64 | 378k | Store(Second, d, out1 + x); |
65 | 378k | Store(Third, d, out2 + x); |
66 | 378k | } |
67 | 974k | } |
68 | 158k | for (; x < w; x++) { |
69 | 72.4k | if (transform_type == 6) { |
70 | 32.1k | pixel_type Y = in0[x]; |
71 | 32.1k | pixel_type Co = in1[x]; |
72 | 32.1k | pixel_type Cg = in2[x]; |
73 | 32.1k | pixel_type tmp = PixelAdd(Y, -(Cg >> 1)); |
74 | 32.1k | pixel_type G = PixelAdd(Cg, tmp); |
75 | 32.1k | pixel_type B = PixelAdd(tmp, -(Co >> 1)); |
76 | 32.1k | pixel_type R = PixelAdd(B, Co); |
77 | 32.1k | out0[x] = R; |
78 | 32.1k | out1[x] = G; |
79 | 32.1k | out2[x] = B; |
80 | 40.2k | } else { |
81 | 40.2k | pixel_type First = in0[x]; |
82 | 40.2k | pixel_type Second = in1[x]; |
83 | 40.2k | pixel_type Third = in2[x]; |
84 | 40.2k | if (third) Third = PixelAdd(Third, First); |
85 | 40.2k | if (second == 1) { |
86 | 24.6k | Second = PixelAdd(Second, First); |
87 | 24.6k | } else if (second == 2) { |
88 | 9.84k | Second = PixelAdd(Second, (PixelAdd(First, Third) >> 1)); |
89 | 9.84k | } |
90 | 40.2k | out0[x] = First; |
91 | 40.2k | out1[x] = Second; |
92 | 40.2k | out2[x] = Third; |
93 | 40.2k | } |
94 | 72.4k | } |
95 | 85.9k | } Unexecuted instantiation: void jxl::N_SSE4::InvRCTRow<0>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE4::InvRCTRow<1>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE4::InvRCTRow<2>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE4::InvRCTRow<3>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE4::InvRCTRow<4>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE4::InvRCTRow<5>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE4::InvRCTRow<6>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_AVX2::InvRCTRow<0>(int const*, int const*, int const*, int*, int*, int*, unsigned long) void jxl::N_AVX2::InvRCTRow<1>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Line | Count | Source | 32 | 2.71k | pixel_type* out2, size_t w) { | 33 | 2.71k | static_assert(transform_type >= 0 && transform_type < 7, | 34 | 2.71k | "Invalid transform type"); | 35 | 2.71k | int second = transform_type >> 1; | 36 | 2.71k | int third = transform_type & 1; | 37 | | | 38 | 2.71k | size_t x = 0; | 39 | 2.71k | const HWY_FULL(pixel_type) d; | 40 | 2.71k | const size_t N = Lanes(d); | 41 | 31.0k | for (; x + N - 1 < w; x += N) { | 42 | 28.3k | if (transform_type == 6) { | 43 | 0 | auto Y = Load(d, in0 + x); | 44 | 0 | auto Co = Load(d, in1 + x); | 45 | 0 | auto Cg = Load(d, in2 + x); | 46 | 0 | Y = Sub(Y, ShiftRight<1>(Cg)); | 47 | 0 | auto G = Add(Cg, Y); | 48 | 0 | Y = Sub(Y, ShiftRight<1>(Co)); | 49 | 0 | auto R = Add(Y, Co); | 50 | 0 | Store(R, d, out0 + x); | 51 | 0 | Store(G, d, out1 + x); | 52 | 0 | Store(Y, d, out2 + x); | 53 | 28.3k | } else { | 54 | 28.3k | auto First = Load(d, in0 + x); | 55 | 28.3k | auto Second = Load(d, in1 + x); | 56 | 28.3k | auto Third = Load(d, in2 + x); | 57 | 28.3k | if (third) Third = Add(Third, First); | 58 | 28.3k | if (second == 1) { | 59 | 0 | Second = Add(Second, First); | 60 | 28.3k | } else if (second == 2) { | 61 | 0 | Second = Add(Second, ShiftRight<1>(Add(First, Third))); | 62 | 0 | } | 63 | 28.3k | Store(First, d, out0 + x); | 64 | 28.3k | Store(Second, d, out1 + x); | 65 | 28.3k | Store(Third, d, out2 + x); | 66 | 28.3k | } | 67 | 28.3k | } | 68 | 8.45k | for (; x < w; x++) { | 69 | 5.73k | if (transform_type == 6) { | 70 | 0 | pixel_type Y = in0[x]; | 71 | 0 | pixel_type Co = in1[x]; | 72 | 0 | pixel_type Cg = in2[x]; | 73 | 0 | pixel_type tmp = PixelAdd(Y, -(Cg >> 1)); | 74 | 0 | pixel_type G = PixelAdd(Cg, tmp); | 75 | 0 | pixel_type B = PixelAdd(tmp, -(Co >> 1)); | 76 | 0 | pixel_type R = PixelAdd(B, Co); | 77 | 0 | out0[x] = R; | 78 | 0 | out1[x] = G; | 79 | 0 | out2[x] = B; | 80 | 5.73k | } else { | 81 | 5.73k | pixel_type First = in0[x]; | 82 | 5.73k | pixel_type Second = in1[x]; | 83 | 5.73k | pixel_type Third = in2[x]; | 84 | 5.73k | if (third) Third = PixelAdd(Third, First); | 85 | 5.73k | if (second == 1) { | 86 | 0 | Second = PixelAdd(Second, First); | 87 | 5.73k | } else if (second == 2) { | 88 | 0 | Second = PixelAdd(Second, (PixelAdd(First, Third) >> 1)); | 89 | 0 | } | 90 | 5.73k | out0[x] = First; | 91 | 5.73k | out1[x] = Second; | 92 | 5.73k | out2[x] = Third; | 93 | 5.73k | } | 94 | 5.73k | } | 95 | 2.71k | } |
void jxl::N_AVX2::InvRCTRow<2>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Line | Count | Source | 32 | 30.8k | pixel_type* out2, size_t w) { | 33 | 30.8k | static_assert(transform_type >= 0 && transform_type < 7, | 34 | 30.8k | "Invalid transform type"); | 35 | 30.8k | int second = transform_type >> 1; | 36 | 30.8k | int third = transform_type & 1; | 37 | | | 38 | 30.8k | size_t x = 0; | 39 | 30.8k | const HWY_FULL(pixel_type) d; | 40 | 30.8k | const size_t N = Lanes(d); | 41 | 331k | for (; x + N - 1 < w; x += N) { | 42 | 300k | if (transform_type == 6) { | 43 | 0 | auto Y = Load(d, in0 + x); | 44 | 0 | auto Co = Load(d, in1 + x); | 45 | 0 | auto Cg = Load(d, in2 + x); | 46 | 0 | Y = Sub(Y, ShiftRight<1>(Cg)); | 47 | 0 | auto G = Add(Cg, Y); | 48 | 0 | Y = Sub(Y, ShiftRight<1>(Co)); | 49 | 0 | auto R = Add(Y, Co); | 50 | 0 | Store(R, d, out0 + x); | 51 | 0 | Store(G, d, out1 + x); | 52 | 0 | Store(Y, d, out2 + x); | 53 | 300k | } else { | 54 | 300k | auto First = Load(d, in0 + x); | 55 | 300k | auto Second = Load(d, in1 + x); | 56 | 300k | auto Third = Load(d, in2 + x); | 57 | 300k | if (third) Third = Add(Third, First); | 58 | 300k | if (second == 1) { | 59 | 300k | Second = Add(Second, First); | 60 | 300k | } else if (second == 2) { | 61 | 0 | Second = Add(Second, ShiftRight<1>(Add(First, Third))); | 62 | 0 | } | 63 | 300k | Store(First, d, out0 + x); | 64 | 300k | Store(Second, d, out1 + x); | 65 | 300k | Store(Third, d, out2 + x); | 66 | 300k | } | 67 | 300k | } | 68 | 48.3k | for (; x < w; x++) { | 69 | 17.5k | if (transform_type == 6) { | 70 | 0 | pixel_type Y = in0[x]; | 71 | 0 | pixel_type Co = in1[x]; | 72 | 0 | pixel_type Cg = in2[x]; | 73 | 0 | pixel_type tmp = PixelAdd(Y, -(Cg >> 1)); | 74 | 0 | pixel_type G = PixelAdd(Cg, tmp); | 75 | 0 | pixel_type B = PixelAdd(tmp, -(Co >> 1)); | 76 | 0 | pixel_type R = PixelAdd(B, Co); | 77 | 0 | out0[x] = R; | 78 | 0 | out1[x] = G; | 79 | 0 | out2[x] = B; | 80 | 17.5k | } else { | 81 | 17.5k | pixel_type First = in0[x]; | 82 | 17.5k | pixel_type Second = in1[x]; | 83 | 17.5k | pixel_type Third = in2[x]; | 84 | 17.5k | if (third) Third = PixelAdd(Third, First); | 85 | 17.5k | if (second == 1) { | 86 | 17.5k | Second = PixelAdd(Second, First); | 87 | 17.5k | } else if (second == 2) { | 88 | 0 | Second = PixelAdd(Second, (PixelAdd(First, Third) >> 1)); | 89 | 0 | } | 90 | 17.5k | out0[x] = First; | 91 | 17.5k | out1[x] = Second; | 92 | 17.5k | out2[x] = Third; | 93 | 17.5k | } | 94 | 17.5k | } | 95 | 30.8k | } |
void jxl::N_AVX2::InvRCTRow<3>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Line | Count | Source | 32 | 3.51k | pixel_type* out2, size_t w) { | 33 | 3.51k | static_assert(transform_type >= 0 && transform_type < 7, | 34 | 3.51k | "Invalid transform type"); | 35 | 3.51k | int second = transform_type >> 1; | 36 | 3.51k | int third = transform_type & 1; | 37 | | | 38 | 3.51k | size_t x = 0; | 39 | 3.51k | const HWY_FULL(pixel_type) d; | 40 | 3.51k | const size_t N = Lanes(d); | 41 | 35.6k | for (; x + N - 1 < w; x += N) { | 42 | 32.1k | if (transform_type == 6) { | 43 | 0 | auto Y = Load(d, in0 + x); | 44 | 0 | auto Co = Load(d, in1 + x); | 45 | 0 | auto Cg = Load(d, in2 + x); | 46 | 0 | Y = Sub(Y, ShiftRight<1>(Cg)); | 47 | 0 | auto G = Add(Cg, Y); | 48 | 0 | Y = Sub(Y, ShiftRight<1>(Co)); | 49 | 0 | auto R = Add(Y, Co); | 50 | 0 | Store(R, d, out0 + x); | 51 | 0 | Store(G, d, out1 + x); | 52 | 0 | Store(Y, d, out2 + x); | 53 | 32.1k | } else { | 54 | 32.1k | auto First = Load(d, in0 + x); | 55 | 32.1k | auto Second = Load(d, in1 + x); | 56 | 32.1k | auto Third = Load(d, in2 + x); | 57 | 32.1k | if (third) Third = Add(Third, First); | 58 | 32.1k | if (second == 1) { | 59 | 32.1k | Second = Add(Second, First); | 60 | 32.1k | } else if (second == 2) { | 61 | 0 | Second = Add(Second, ShiftRight<1>(Add(First, Third))); | 62 | 0 | } | 63 | 32.1k | Store(First, d, out0 + x); | 64 | 32.1k | Store(Second, d, out1 + x); | 65 | 32.1k | Store(Third, d, out2 + x); | 66 | 32.1k | } | 67 | 32.1k | } | 68 | 10.6k | for (; x < w; x++) { | 69 | 7.13k | if (transform_type == 6) { | 70 | 0 | pixel_type Y = in0[x]; | 71 | 0 | pixel_type Co = in1[x]; | 72 | 0 | pixel_type Cg = in2[x]; | 73 | 0 | pixel_type tmp = PixelAdd(Y, -(Cg >> 1)); | 74 | 0 | pixel_type G = PixelAdd(Cg, tmp); | 75 | 0 | pixel_type B = PixelAdd(tmp, -(Co >> 1)); | 76 | 0 | pixel_type R = PixelAdd(B, Co); | 77 | 0 | out0[x] = R; | 78 | 0 | out1[x] = G; | 79 | 0 | out2[x] = B; | 80 | 7.13k | } else { | 81 | 7.13k | pixel_type First = in0[x]; | 82 | 7.13k | pixel_type Second = in1[x]; | 83 | 7.13k | pixel_type Third = in2[x]; | 84 | 7.13k | if (third) Third = PixelAdd(Third, First); | 85 | 7.13k | if (second == 1) { | 86 | 7.13k | Second = PixelAdd(Second, First); | 87 | 7.13k | } else if (second == 2) { | 88 | 0 | Second = PixelAdd(Second, (PixelAdd(First, Third) >> 1)); | 89 | 0 | } | 90 | 7.13k | out0[x] = First; | 91 | 7.13k | out1[x] = Second; | 92 | 7.13k | out2[x] = Third; | 93 | 7.13k | } | 94 | 7.13k | } | 95 | 3.51k | } |
void jxl::N_AVX2::InvRCTRow<4>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Line | Count | Source | 32 | 2.63k | pixel_type* out2, size_t w) { | 33 | 2.63k | static_assert(transform_type >= 0 && transform_type < 7, | 34 | 2.63k | "Invalid transform type"); | 35 | 2.63k | int second = transform_type >> 1; | 36 | 2.63k | int third = transform_type & 1; | 37 | | | 38 | 2.63k | size_t x = 0; | 39 | 2.63k | const HWY_FULL(pixel_type) d; | 40 | 2.63k | const size_t N = Lanes(d); | 41 | 16.3k | for (; x + N - 1 < w; x += N) { | 42 | 13.6k | if (transform_type == 6) { | 43 | 0 | auto Y = Load(d, in0 + x); | 44 | 0 | auto Co = Load(d, in1 + x); | 45 | 0 | auto Cg = Load(d, in2 + x); | 46 | 0 | Y = Sub(Y, ShiftRight<1>(Cg)); | 47 | 0 | auto G = Add(Cg, Y); | 48 | 0 | Y = Sub(Y, ShiftRight<1>(Co)); | 49 | 0 | auto R = Add(Y, Co); | 50 | 0 | Store(R, d, out0 + x); | 51 | 0 | Store(G, d, out1 + x); | 52 | 0 | Store(Y, d, out2 + x); | 53 | 13.6k | } else { | 54 | 13.6k | auto First = Load(d, in0 + x); | 55 | 13.6k | auto Second = Load(d, in1 + x); | 56 | 13.6k | auto Third = Load(d, in2 + x); | 57 | 13.6k | if (third) Third = Add(Third, First); | 58 | 13.6k | if (second == 1) { | 59 | 0 | Second = Add(Second, First); | 60 | 13.6k | } else if (second == 2) { | 61 | 13.6k | Second = Add(Second, ShiftRight<1>(Add(First, Third))); | 62 | 13.6k | } | 63 | 13.6k | Store(First, d, out0 + x); | 64 | 13.6k | Store(Second, d, out1 + x); | 65 | 13.6k | Store(Third, d, out2 + x); | 66 | 13.6k | } | 67 | 13.6k | } | 68 | 11.4k | for (; x < w; x++) { | 69 | 8.77k | if (transform_type == 6) { | 70 | 0 | pixel_type Y = in0[x]; | 71 | 0 | pixel_type Co = in1[x]; | 72 | 0 | pixel_type Cg = in2[x]; | 73 | 0 | pixel_type tmp = PixelAdd(Y, -(Cg >> 1)); | 74 | 0 | pixel_type G = PixelAdd(Cg, tmp); | 75 | 0 | pixel_type B = PixelAdd(tmp, -(Co >> 1)); | 76 | 0 | pixel_type R = PixelAdd(B, Co); | 77 | 0 | out0[x] = R; | 78 | 0 | out1[x] = G; | 79 | 0 | out2[x] = B; | 80 | 8.77k | } else { | 81 | 8.77k | pixel_type First = in0[x]; | 82 | 8.77k | pixel_type Second = in1[x]; | 83 | 8.77k | pixel_type Third = in2[x]; | 84 | 8.77k | if (third) Third = PixelAdd(Third, First); | 85 | 8.77k | if (second == 1) { | 86 | 0 | Second = PixelAdd(Second, First); | 87 | 8.77k | } else if (second == 2) { | 88 | 8.77k | Second = PixelAdd(Second, (PixelAdd(First, Third) >> 1)); | 89 | 8.77k | } | 90 | 8.77k | out0[x] = First; | 91 | 8.77k | out1[x] = Second; | 92 | 8.77k | out2[x] = Third; | 93 | 8.77k | } | 94 | 8.77k | } | 95 | 2.63k | } |
void jxl::N_AVX2::InvRCTRow<5>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Line | Count | Source | 32 | 569 | pixel_type* out2, size_t w) { | 33 | 569 | static_assert(transform_type >= 0 && transform_type < 7, | 34 | 569 | "Invalid transform type"); | 35 | 569 | int second = transform_type >> 1; | 36 | 569 | int third = transform_type & 1; | 37 | | | 38 | 569 | size_t x = 0; | 39 | 569 | const HWY_FULL(pixel_type) d; | 40 | 569 | const size_t N = Lanes(d); | 41 | 3.92k | for (; x + N - 1 < w; x += N) { | 42 | 3.35k | if (transform_type == 6) { | 43 | 0 | auto Y = Load(d, in0 + x); | 44 | 0 | auto Co = Load(d, in1 + x); | 45 | 0 | auto Cg = Load(d, in2 + x); | 46 | 0 | Y = Sub(Y, ShiftRight<1>(Cg)); | 47 | 0 | auto G = Add(Cg, Y); | 48 | 0 | Y = Sub(Y, ShiftRight<1>(Co)); | 49 | 0 | auto R = Add(Y, Co); | 50 | 0 | Store(R, d, out0 + x); | 51 | 0 | Store(G, d, out1 + x); | 52 | 0 | Store(Y, d, out2 + x); | 53 | 3.35k | } else { | 54 | 3.35k | auto First = Load(d, in0 + x); | 55 | 3.35k | auto Second = Load(d, in1 + x); | 56 | 3.35k | auto Third = Load(d, in2 + x); | 57 | 3.35k | if (third) Third = Add(Third, First); | 58 | 3.35k | if (second == 1) { | 59 | 0 | Second = Add(Second, First); | 60 | 3.35k | } else if (second == 2) { | 61 | 3.35k | Second = Add(Second, ShiftRight<1>(Add(First, Third))); | 62 | 3.35k | } | 63 | 3.35k | Store(First, d, out0 + x); | 64 | 3.35k | Store(Second, d, out1 + x); | 65 | 3.35k | Store(Third, d, out2 + x); | 66 | 3.35k | } | 67 | 3.35k | } | 68 | 1.63k | for (; x < w; x++) { | 69 | 1.06k | if (transform_type == 6) { | 70 | 0 | pixel_type Y = in0[x]; | 71 | 0 | pixel_type Co = in1[x]; | 72 | 0 | pixel_type Cg = in2[x]; | 73 | 0 | pixel_type tmp = PixelAdd(Y, -(Cg >> 1)); | 74 | 0 | pixel_type G = PixelAdd(Cg, tmp); | 75 | 0 | pixel_type B = PixelAdd(tmp, -(Co >> 1)); | 76 | 0 | pixel_type R = PixelAdd(B, Co); | 77 | 0 | out0[x] = R; | 78 | 0 | out1[x] = G; | 79 | 0 | out2[x] = B; | 80 | 1.06k | } else { | 81 | 1.06k | pixel_type First = in0[x]; | 82 | 1.06k | pixel_type Second = in1[x]; | 83 | 1.06k | pixel_type Third = in2[x]; | 84 | 1.06k | if (third) Third = PixelAdd(Third, First); | 85 | 1.06k | if (second == 1) { | 86 | 0 | Second = PixelAdd(Second, First); | 87 | 1.06k | } else if (second == 2) { | 88 | 1.06k | Second = PixelAdd(Second, (PixelAdd(First, Third) >> 1)); | 89 | 1.06k | } | 90 | 1.06k | out0[x] = First; | 91 | 1.06k | out1[x] = Second; | 92 | 1.06k | out2[x] = Third; | 93 | 1.06k | } | 94 | 1.06k | } | 95 | 569 | } |
void jxl::N_AVX2::InvRCTRow<6>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Line | Count | Source | 32 | 45.7k | pixel_type* out2, size_t w) { | 33 | 45.7k | static_assert(transform_type >= 0 && transform_type < 7, | 34 | 45.7k | "Invalid transform type"); | 35 | 45.7k | int second = transform_type >> 1; | 36 | 45.7k | int third = transform_type & 1; | 37 | | | 38 | 45.7k | size_t x = 0; | 39 | 45.7k | const HWY_FULL(pixel_type) d; | 40 | 45.7k | const size_t N = Lanes(d); | 41 | 641k | for (; x + N - 1 < w; x += N) { | 42 | 595k | if (transform_type == 6) { | 43 | 595k | auto Y = Load(d, in0 + x); | 44 | 595k | auto Co = Load(d, in1 + x); | 45 | 595k | auto Cg = Load(d, in2 + x); | 46 | 595k | Y = Sub(Y, ShiftRight<1>(Cg)); | 47 | 595k | auto G = Add(Cg, Y); | 48 | 595k | Y = Sub(Y, ShiftRight<1>(Co)); | 49 | 595k | auto R = Add(Y, Co); | 50 | 595k | Store(R, d, out0 + x); | 51 | 595k | Store(G, d, out1 + x); | 52 | 595k | Store(Y, d, out2 + x); | 53 | 595k | } else { | 54 | 0 | auto First = Load(d, in0 + x); | 55 | 0 | auto Second = Load(d, in1 + x); | 56 | 0 | auto Third = Load(d, in2 + x); | 57 | 0 | if (third) Third = Add(Third, First); | 58 | 0 | if (second == 1) { | 59 | 0 | Second = Add(Second, First); | 60 | 0 | } else if (second == 2) { | 61 | 0 | Second = Add(Second, ShiftRight<1>(Add(First, Third))); | 62 | 0 | } | 63 | 0 | Store(First, d, out0 + x); | 64 | 0 | Store(Second, d, out1 + x); | 65 | 0 | Store(Third, d, out2 + x); | 66 | 0 | } | 67 | 595k | } | 68 | 77.8k | for (; x < w; x++) { | 69 | 32.1k | if (transform_type == 6) { | 70 | 32.1k | pixel_type Y = in0[x]; | 71 | 32.1k | pixel_type Co = in1[x]; | 72 | 32.1k | pixel_type Cg = in2[x]; | 73 | 32.1k | pixel_type tmp = PixelAdd(Y, -(Cg >> 1)); | 74 | 32.1k | pixel_type G = PixelAdd(Cg, tmp); | 75 | 32.1k | pixel_type B = PixelAdd(tmp, -(Co >> 1)); | 76 | 32.1k | pixel_type R = PixelAdd(B, Co); | 77 | 32.1k | out0[x] = R; | 78 | 32.1k | out1[x] = G; | 79 | 32.1k | out2[x] = B; | 80 | 32.1k | } else { | 81 | 0 | pixel_type First = in0[x]; | 82 | 0 | pixel_type Second = in1[x]; | 83 | 0 | pixel_type Third = in2[x]; | 84 | 0 | if (third) Third = PixelAdd(Third, First); | 85 | 0 | if (second == 1) { | 86 | 0 | Second = PixelAdd(Second, First); | 87 | 0 | } else if (second == 2) { | 88 | 0 | Second = PixelAdd(Second, (PixelAdd(First, Third) >> 1)); | 89 | 0 | } | 90 | 0 | out0[x] = First; | 91 | 0 | out1[x] = Second; | 92 | 0 | out2[x] = Third; | 93 | 0 | } | 94 | 32.1k | } | 95 | 45.7k | } |
Unexecuted instantiation: void jxl::N_SSE2::InvRCTRow<0>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE2::InvRCTRow<1>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE2::InvRCTRow<2>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE2::InvRCTRow<3>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE2::InvRCTRow<4>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE2::InvRCTRow<5>(int const*, int const*, int const*, int*, int*, int*, unsigned long) Unexecuted instantiation: void jxl::N_SSE2::InvRCTRow<6>(int const*, int const*, int const*, int*, int*, int*, unsigned long) |
96 | | |
97 | 3.94k | Status InvRCT(Image& input, size_t begin_c, size_t rct_type, ThreadPool* pool) { |
98 | 3.94k | JXL_RETURN_IF_ERROR(CheckEqualChannels(input, begin_c, begin_c + 2)); |
99 | 3.94k | size_t m = begin_c; |
100 | 3.94k | Channel& c0 = input.channel[m + 0]; |
101 | 3.94k | size_t w = c0.w; |
102 | 3.94k | size_t h = c0.h; |
103 | 3.94k | if (rct_type == 0) { // noop |
104 | 1.60k | return true; |
105 | 1.60k | } |
106 | | // Permutation: 0=RGB, 1=GBR, 2=BRG, 3=RBG, 4=GRB, 5=BGR |
107 | 2.34k | int permutation = rct_type / 7; |
108 | 2.34k | JXL_ENSURE(permutation < 6); |
109 | | // 0-5 values have the low bit corresponding to Third and the high bits |
110 | | // corresponding to Second. 6 corresponds to YCoCg. |
111 | | // |
112 | | // Second: 0=nop, 1=SubtractFirst, 2=SubtractAvgFirstThird |
113 | | // |
114 | | // Third: 0=nop, 1=SubtractFirst |
115 | 2.34k | int custom = rct_type % 7; |
116 | | // Special case: permute-only. Swap channels around. |
117 | 2.34k | if (custom == 0) { |
118 | 29 | Channel ch0 = std::move(input.channel[m]); |
119 | 29 | Channel ch1 = std::move(input.channel[m + 1]); |
120 | 29 | Channel ch2 = std::move(input.channel[m + 2]); |
121 | 29 | input.channel[m + (permutation % 3)] = std::move(ch0); |
122 | 29 | input.channel[m + ((permutation + 1 + permutation / 3) % 3)] = |
123 | 29 | std::move(ch1); |
124 | 29 | input.channel[m + ((permutation + 2 - permutation / 3) % 3)] = |
125 | 29 | std::move(ch2); |
126 | 29 | return true; |
127 | 29 | } |
128 | 2.31k | constexpr decltype(&InvRCTRow<0>) inv_rct_row[] = { |
129 | 2.31k | InvRCTRow<0>, InvRCTRow<1>, InvRCTRow<2>, InvRCTRow<3>, |
130 | 2.31k | InvRCTRow<4>, InvRCTRow<5>, InvRCTRow<6>}; |
131 | 2.31k | const auto process_row = [&](const uint32_t task, |
132 | 85.9k | size_t /* thread */) -> Status { |
133 | 85.9k | const size_t y = task; |
134 | 85.9k | const pixel_type* in0 = input.channel[m].Row(y); |
135 | 85.9k | const pixel_type* in1 = input.channel[m + 1].Row(y); |
136 | 85.9k | const pixel_type* in2 = input.channel[m + 2].Row(y); |
137 | 85.9k | pixel_type* out0 = input.channel[m + (permutation % 3)].Row(y); |
138 | 85.9k | pixel_type* out1 = |
139 | 85.9k | input.channel[m + ((permutation + 1 + permutation / 3) % 3)].Row(y); |
140 | 85.9k | pixel_type* out2 = |
141 | 85.9k | input.channel[m + ((permutation + 2 - permutation / 3) % 3)].Row(y); |
142 | 85.9k | inv_rct_row[custom](in0, in1, in2, out0, out1, out2, w); |
143 | 85.9k | return true; |
144 | 85.9k | }; Unexecuted instantiation: rct.cc:jxl::N_SSE4::InvRCT(jxl::Image&, unsigned long, unsigned long, jxl::ThreadPool*)::$_0::operator()(unsigned int, unsigned long) const rct.cc:jxl::N_AVX2::InvRCT(jxl::Image&, unsigned long, unsigned long, jxl::ThreadPool*)::$_0::operator()(unsigned int, unsigned long) const Line | Count | Source | 132 | 85.9k | size_t /* thread */) -> Status { | 133 | 85.9k | const size_t y = task; | 134 | 85.9k | const pixel_type* in0 = input.channel[m].Row(y); | 135 | 85.9k | const pixel_type* in1 = input.channel[m + 1].Row(y); | 136 | 85.9k | const pixel_type* in2 = input.channel[m + 2].Row(y); | 137 | 85.9k | pixel_type* out0 = input.channel[m + (permutation % 3)].Row(y); | 138 | 85.9k | pixel_type* out1 = | 139 | 85.9k | input.channel[m + ((permutation + 1 + permutation / 3) % 3)].Row(y); | 140 | 85.9k | pixel_type* out2 = | 141 | 85.9k | input.channel[m + ((permutation + 2 - permutation / 3) % 3)].Row(y); | 142 | 85.9k | inv_rct_row[custom](in0, in1, in2, out0, out1, out2, w); | 143 | 85.9k | return true; | 144 | 85.9k | }; |
Unexecuted instantiation: rct.cc:jxl::N_SSE2::InvRCT(jxl::Image&, unsigned long, unsigned long, jxl::ThreadPool*)::$_0::operator()(unsigned int, unsigned long) const |
145 | 2.31k | JXL_RETURN_IF_ERROR( |
146 | 2.31k | RunOnPool(pool, 0, h, ThreadPool::NoInit, process_row, "InvRCT")); |
147 | 2.31k | return true; |
148 | 2.31k | } Unexecuted instantiation: jxl::N_SSE4::InvRCT(jxl::Image&, unsigned long, unsigned long, jxl::ThreadPool*) jxl::N_AVX2::InvRCT(jxl::Image&, unsigned long, unsigned long, jxl::ThreadPool*) Line | Count | Source | 97 | 3.94k | Status InvRCT(Image& input, size_t begin_c, size_t rct_type, ThreadPool* pool) { | 98 | 3.94k | JXL_RETURN_IF_ERROR(CheckEqualChannels(input, begin_c, begin_c + 2)); | 99 | 3.94k | size_t m = begin_c; | 100 | 3.94k | Channel& c0 = input.channel[m + 0]; | 101 | 3.94k | size_t w = c0.w; | 102 | 3.94k | size_t h = c0.h; | 103 | 3.94k | if (rct_type == 0) { // noop | 104 | 1.60k | return true; | 105 | 1.60k | } | 106 | | // Permutation: 0=RGB, 1=GBR, 2=BRG, 3=RBG, 4=GRB, 5=BGR | 107 | 2.34k | int permutation = rct_type / 7; | 108 | 2.34k | JXL_ENSURE(permutation < 6); | 109 | | // 0-5 values have the low bit corresponding to Third and the high bits | 110 | | // corresponding to Second. 6 corresponds to YCoCg. | 111 | | // | 112 | | // Second: 0=nop, 1=SubtractFirst, 2=SubtractAvgFirstThird | 113 | | // | 114 | | // Third: 0=nop, 1=SubtractFirst | 115 | 2.34k | int custom = rct_type % 7; | 116 | | // Special case: permute-only. Swap channels around. | 117 | 2.34k | if (custom == 0) { | 118 | 29 | Channel ch0 = std::move(input.channel[m]); | 119 | 29 | Channel ch1 = std::move(input.channel[m + 1]); | 120 | 29 | Channel ch2 = std::move(input.channel[m + 2]); | 121 | 29 | input.channel[m + (permutation % 3)] = std::move(ch0); | 122 | 29 | input.channel[m + ((permutation + 1 + permutation / 3) % 3)] = | 123 | 29 | std::move(ch1); | 124 | 29 | input.channel[m + ((permutation + 2 - permutation / 3) % 3)] = | 125 | 29 | std::move(ch2); | 126 | 29 | return true; | 127 | 29 | } | 128 | 2.31k | constexpr decltype(&InvRCTRow<0>) inv_rct_row[] = { | 129 | 2.31k | InvRCTRow<0>, InvRCTRow<1>, InvRCTRow<2>, InvRCTRow<3>, | 130 | 2.31k | InvRCTRow<4>, InvRCTRow<5>, InvRCTRow<6>}; | 131 | 2.31k | const auto process_row = [&](const uint32_t task, | 132 | 2.31k | size_t /* thread */) -> Status { | 133 | 2.31k | const size_t y = task; | 134 | 2.31k | const pixel_type* in0 = input.channel[m].Row(y); | 135 | 2.31k | const pixel_type* in1 = input.channel[m + 1].Row(y); | 136 | 2.31k | const pixel_type* in2 = input.channel[m + 2].Row(y); | 137 | 2.31k | pixel_type* out0 = input.channel[m + (permutation % 3)].Row(y); | 138 | 2.31k | pixel_type* out1 = | 139 | 2.31k | input.channel[m + ((permutation + 1 + permutation / 3) % 3)].Row(y); | 140 | 2.31k | pixel_type* out2 = | 141 | 2.31k | input.channel[m + ((permutation + 2 - permutation / 3) % 3)].Row(y); | 142 | 2.31k | inv_rct_row[custom](in0, in1, in2, out0, out1, out2, w); | 143 | 2.31k | return true; | 144 | 2.31k | }; | 145 | 2.31k | JXL_RETURN_IF_ERROR( | 146 | 2.31k | RunOnPool(pool, 0, h, ThreadPool::NoInit, process_row, "InvRCT")); | 147 | 2.31k | return true; | 148 | 2.31k | } |
Unexecuted instantiation: jxl::N_SSE2::InvRCT(jxl::Image&, unsigned long, unsigned long, jxl::ThreadPool*) |
149 | | |
150 | | } // namespace HWY_NAMESPACE |
151 | | } // namespace jxl |
152 | | HWY_AFTER_NAMESPACE(); |
153 | | |
154 | | #if HWY_ONCE |
155 | | namespace jxl { |
156 | | |
157 | | HWY_EXPORT(InvRCT); |
158 | 3.94k | Status InvRCT(Image& input, size_t begin_c, size_t rct_type, ThreadPool* pool) { |
159 | 3.94k | return HWY_DYNAMIC_DISPATCH(InvRCT)(input, begin_c, rct_type, pool); |
160 | 3.94k | } |
161 | | |
162 | | } // namespace jxl |
163 | | #endif |