/work/dav1d/src/cdef_tmpl.c
Line | Count | Source |
1 | | /* |
2 | | * Copyright © 2018, VideoLAN and dav1d authors |
3 | | * Copyright © 2018, Two Orioles, LLC |
4 | | * All rights reserved. |
5 | | * |
6 | | * Redistribution and use in source and binary forms, with or without |
7 | | * modification, are permitted provided that the following conditions are met: |
8 | | * |
9 | | * 1. Redistributions of source code must retain the above copyright notice, this |
10 | | * list of conditions and the following disclaimer. |
11 | | * |
12 | | * 2. Redistributions in binary form must reproduce the above copyright notice, |
13 | | * this list of conditions and the following disclaimer in the documentation |
14 | | * and/or other materials provided with the distribution. |
15 | | * |
16 | | * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND |
17 | | * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED |
18 | | * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE |
19 | | * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR |
20 | | * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES |
21 | | * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; |
22 | | * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND |
23 | | * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT |
24 | | * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS |
25 | | * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. |
26 | | */ |
27 | | |
28 | | #include "config.h" |
29 | | |
30 | | #include <stdlib.h> |
31 | | |
32 | | #include "common/intops.h" |
33 | | |
34 | | #include "src/cdef.h" |
35 | | #include "src/tables.h" |
36 | | |
37 | | static inline int constrain(const int diff, const int threshold, |
38 | | const int shift) |
39 | 2.52G | { |
40 | 2.52G | const int adiff = abs(diff); |
41 | 2.52G | return apply_sign(imin(adiff, imax(0, threshold - (adiff >> shift))), diff); |
42 | 2.52G | } |
43 | | |
44 | | static inline void fill(int16_t *tmp, const ptrdiff_t stride, |
45 | | const int w, const int h) |
46 | 3.47M | { |
47 | | /* Use a value that's a large positive number when interpreted as unsigned, |
48 | | * and a large negative number when interpreted as signed. */ |
49 | 26.2M | for (int y = 0; y < h; y++) { |
50 | 94.5M | for (int x = 0; x < w; x++) |
51 | 71.7M | tmp[x] = INT16_MIN; |
52 | 22.8M | tmp += stride; |
53 | 22.8M | } |
54 | 3.47M | } |
55 | | |
56 | | static void padding(int16_t *tmp, const ptrdiff_t tmp_stride, |
57 | | const pixel *src, const ptrdiff_t src_stride, |
58 | | const pixel (*left)[2], |
59 | | const pixel *top, const pixel *bottom, |
60 | | const int w, const int h, const enum CdefEdgeFlags edges) |
61 | 11.6M | { |
62 | | // fill extended input buffer |
63 | 11.6M | int x_start = -2, x_end = w + 2, y_start = -2, y_end = h + 2; |
64 | 11.6M | if (!(edges & CDEF_HAVE_TOP)) { |
65 | 802k | fill(tmp - 2 - 2 * tmp_stride, tmp_stride, w + 4, 2); |
66 | 802k | y_start = 0; |
67 | 802k | } |
68 | 11.6M | if (!(edges & CDEF_HAVE_BOTTOM)) { |
69 | 749k | fill(tmp + h * tmp_stride - 2, tmp_stride, w + 4, 2); |
70 | 749k | y_end -= 2; |
71 | 749k | } |
72 | 11.6M | if (!(edges & CDEF_HAVE_LEFT)) { |
73 | 963k | fill(tmp + y_start * tmp_stride - 2, tmp_stride, 2, y_end - y_start); |
74 | 963k | x_start = 0; |
75 | 963k | } |
76 | 11.6M | if (!(edges & CDEF_HAVE_RIGHT)) { |
77 | 960k | fill(tmp + y_start * tmp_stride + w, tmp_stride, 2, y_end - y_start); |
78 | 960k | x_end -= 2; |
79 | 960k | } |
80 | | |
81 | 33.4M | for (int y = y_start; y < 0; y++) { |
82 | 246M | for (int x = x_start; x < x_end; x++) |
83 | 225M | tmp[x + y * tmp_stride] = top[x]; |
84 | 21.7M | top += PXSTRIDE(src_stride); |
85 | 21.7M | } |
86 | 89.8M | for (int y = 0; y < h; y++) |
87 | 221M | for (int x = x_start; x < 0; x++) |
88 | 143M | tmp[x + y * tmp_stride] = left[y][2 + x]; |
89 | 87.6M | for (int y = 0; y < h; y++) { |
90 | 18.4E | for (int x = (y < h) ? 0 : x_start; x < x_end; x++) |
91 | 683M | tmp[x] = src[x]; |
92 | 76.0M | src += PXSTRIDE(src_stride); |
93 | 76.0M | tmp += tmp_stride; |
94 | 76.0M | } |
95 | 33.4M | for (int y = h; y < y_end; y++) { |
96 | 244M | for (int x = x_start; x < x_end; x++) |
97 | 222M | tmp[x] = bottom[x]; |
98 | 21.7M | bottom += PXSTRIDE(src_stride); |
99 | 21.7M | tmp += tmp_stride; |
100 | 21.7M | } |
101 | | |
102 | 11.6M | } |
103 | | |
104 | | static NOINLINE void |
105 | | cdef_filter_block_c(pixel *dst, const ptrdiff_t dst_stride, |
106 | | const pixel (*left)[2], |
107 | | const pixel *const top, const pixel *const bottom, |
108 | | const int pri_strength, const int sec_strength, |
109 | | const int dir, const int damping, const int w, int h, |
110 | | const enum CdefEdgeFlags edges HIGHBD_DECL_SUFFIX) |
111 | 11.6M | { |
112 | 11.6M | const ptrdiff_t tmp_stride = 12; |
113 | 11.6M | assert((w == 4 || w == 8) && (h == 4 || h == 8)); |
114 | 11.6M | int16_t tmp_buf[144]; // 12*12 is the maximum value of tmp_stride * (h + 4) |
115 | 11.6M | int16_t *tmp = tmp_buf + 2 * tmp_stride + 2; |
116 | 11.6M | const int8_t (*const cdef_dirs)[2] = &dav1d_cdef_directions[dir]; |
117 | | |
118 | 11.6M | padding(tmp, tmp_stride, dst, dst_stride, left, top, bottom, w, h, edges); |
119 | | |
120 | 11.6M | if (pri_strength) { |
121 | 9.27M | const int bitdepth_min_8 = bitdepth_from_max(bitdepth_max) - 8; |
122 | 9.27M | const int pri_tap = 4 - ((pri_strength >> bitdepth_min_8) & 1); |
123 | 9.27M | const int pri_shift = imax(0, damping - ulog2(pri_strength)); |
124 | 9.27M | if (sec_strength) { |
125 | 6.80M | const int sec_shift = damping - ulog2(sec_strength); |
126 | 39.7M | do { |
127 | 215M | for (int x = 0; x < w; x++) { |
128 | 176M | const int px = dst[x]; |
129 | 176M | int sum = 0; |
130 | 176M | int max = px, min = px; |
131 | 176M | int pri_tap_k = pri_tap; |
132 | 486M | for (int k = 0; k < 2; k++) { |
133 | 310M | const int off1 = cdef_dirs[2][k]; // dir |
134 | 310M | const int p0 = tmp[x + off1]; |
135 | 310M | const int p1 = tmp[x - off1]; |
136 | 310M | sum += pri_tap_k * constrain(p0 - px, pri_strength, pri_shift); |
137 | 310M | sum += pri_tap_k * constrain(p1 - px, pri_strength, pri_shift); |
138 | | // if pri_tap_k == 4 then it becomes 2 else it remains 3 |
139 | 310M | pri_tap_k = (pri_tap_k & 3) | 2; |
140 | 310M | min = umin(p0, min); |
141 | 310M | max = imax(p0, max); |
142 | 310M | min = umin(p1, min); |
143 | 310M | max = imax(p1, max); |
144 | 310M | const int off2 = cdef_dirs[4][k]; // dir + 2 |
145 | 310M | const int off3 = cdef_dirs[0][k]; // dir - 2 |
146 | 310M | const int s0 = tmp[x + off2]; |
147 | 310M | const int s1 = tmp[x - off2]; |
148 | 310M | const int s2 = tmp[x + off3]; |
149 | 310M | const int s3 = tmp[x - off3]; |
150 | | // sec_tap starts at 2 and becomes 1 |
151 | 310M | const int sec_tap = 2 - k; |
152 | 310M | sum += sec_tap * constrain(s0 - px, sec_strength, sec_shift); |
153 | 310M | sum += sec_tap * constrain(s1 - px, sec_strength, sec_shift); |
154 | 310M | sum += sec_tap * constrain(s2 - px, sec_strength, sec_shift); |
155 | 310M | sum += sec_tap * constrain(s3 - px, sec_strength, sec_shift); |
156 | 310M | min = umin(s0, min); |
157 | 310M | max = imax(s0, max); |
158 | 310M | min = umin(s1, min); |
159 | 310M | max = imax(s1, max); |
160 | 310M | min = umin(s2, min); |
161 | 310M | max = imax(s2, max); |
162 | 310M | min = umin(s3, min); |
163 | 310M | max = imax(s3, max); |
164 | 310M | } |
165 | 176M | dst[x] = iclip(px + ((sum - (sum < 0) + 8) >> 4), min, max); |
166 | 176M | } |
167 | 39.7M | dst += PXSTRIDE(dst_stride); |
168 | 39.7M | tmp += tmp_stride; |
169 | 39.7M | } while (--h); |
170 | 6.80M | } else { // pri_strength only |
171 | 13.5M | do { |
172 | 90.4M | for (int x = 0; x < w; x++) { |
173 | 76.9M | const int px = dst[x]; |
174 | 76.9M | int sum = 0; |
175 | 76.9M | int pri_tap_k = pri_tap; |
176 | 227M | for (int k = 0; k < 2; k++) { |
177 | 150M | const int off = cdef_dirs[2][k]; // dir |
178 | 150M | const int p0 = tmp[x + off]; |
179 | 150M | const int p1 = tmp[x - off]; |
180 | 150M | sum += pri_tap_k * constrain(p0 - px, pri_strength, pri_shift); |
181 | 150M | sum += pri_tap_k * constrain(p1 - px, pri_strength, pri_shift); |
182 | 150M | pri_tap_k = (pri_tap_k & 3) | 2; |
183 | 150M | } |
184 | 76.9M | dst[x] = px + ((sum - (sum < 0) + 8) >> 4); |
185 | 76.9M | } |
186 | 13.5M | dst += PXSTRIDE(dst_stride); |
187 | 13.5M | tmp += tmp_stride; |
188 | 13.5M | } while (--h); |
189 | 2.47M | } |
190 | 9.27M | } else { // sec_strength only |
191 | 2.39M | assert(sec_strength); |
192 | 2.39M | const int sec_shift = damping - ulog2(sec_strength); |
193 | 16.2M | do { |
194 | 109M | for (int x = 0; x < w; x++) { |
195 | 92.9M | const int px = dst[x]; |
196 | 92.9M | int sum = 0; |
197 | 273M | for (int k = 0; k < 2; k++) { |
198 | 180M | const int off1 = cdef_dirs[4][k]; // dir + 2 |
199 | 180M | const int off2 = cdef_dirs[0][k]; // dir - 2 |
200 | 180M | const int s0 = tmp[x + off1]; |
201 | 180M | const int s1 = tmp[x - off1]; |
202 | 180M | const int s2 = tmp[x + off2]; |
203 | 180M | const int s3 = tmp[x - off2]; |
204 | 180M | const int sec_tap = 2 - k; |
205 | 180M | sum += sec_tap * constrain(s0 - px, sec_strength, sec_shift); |
206 | 180M | sum += sec_tap * constrain(s1 - px, sec_strength, sec_shift); |
207 | 180M | sum += sec_tap * constrain(s2 - px, sec_strength, sec_shift); |
208 | 180M | sum += sec_tap * constrain(s3 - px, sec_strength, sec_shift); |
209 | 180M | } |
210 | 92.9M | dst[x] = px + ((sum - (sum < 0) + 8) >> 4); |
211 | 92.9M | } |
212 | 16.2M | dst += PXSTRIDE(dst_stride); |
213 | 16.2M | tmp += tmp_stride; |
214 | 16.2M | } while (--h); |
215 | 2.39M | } |
216 | 11.6M | } |
217 | | |
218 | | #define cdef_fn(w, h) \ |
219 | | static void cdef_filter_block_##w##x##h##_c(pixel *const dst, \ |
220 | | const ptrdiff_t stride, \ |
221 | | const pixel (*left)[2], \ |
222 | | const pixel *const top, \ |
223 | | const pixel *const bottom, \ |
224 | | const int pri_strength, \ |
225 | | const int sec_strength, \ |
226 | | const int dir, \ |
227 | | const int damping, \ |
228 | | const enum CdefEdgeFlags edges \ |
229 | 11.6M | HIGHBD_DECL_SUFFIX) \ |
230 | 11.6M | { \ |
231 | 11.6M | cdef_filter_block_c(dst, stride, left, top, bottom, \ |
232 | 11.6M | pri_strength, sec_strength, dir, damping, w, h, edges HIGHBD_TAIL_SUFFIX); \ |
233 | 11.6M | } cdef_tmpl.c:cdef_filter_block_8x8_c Line | Count | Source | 229 | 7.77M | HIGHBD_DECL_SUFFIX) \ | 230 | 7.77M | { \ | 231 | 7.77M | cdef_filter_block_c(dst, stride, left, top, bottom, \ | 232 | 7.77M | pri_strength, sec_strength, dir, damping, w, h, edges HIGHBD_TAIL_SUFFIX); \ | 233 | 7.77M | } |
cdef_tmpl.c:cdef_filter_block_4x8_c Line | Count | Source | 229 | 57.9k | HIGHBD_DECL_SUFFIX) \ | 230 | 57.9k | { \ | 231 | 57.9k | cdef_filter_block_c(dst, stride, left, top, bottom, \ | 232 | 57.9k | pri_strength, sec_strength, dir, damping, w, h, edges HIGHBD_TAIL_SUFFIX); \ | 233 | 57.9k | } |
cdef_tmpl.c:cdef_filter_block_4x4_c Line | Count | Source | 229 | 3.78M | HIGHBD_DECL_SUFFIX) \ | 230 | 3.78M | { \ | 231 | 3.78M | cdef_filter_block_c(dst, stride, left, top, bottom, \ | 232 | 3.78M | pri_strength, sec_strength, dir, damping, w, h, edges HIGHBD_TAIL_SUFFIX); \ | 233 | 3.78M | } |
|
234 | | |
235 | | cdef_fn(4, 4); |
236 | | cdef_fn(4, 8); |
237 | | cdef_fn(8, 8); |
238 | | |
239 | | static int cdef_find_dir_c(const pixel *img, const ptrdiff_t stride, |
240 | | unsigned *const var HIGHBD_DECL_SUFFIX) |
241 | 5.06M | { |
242 | 5.06M | const int bitdepth_min_8 = bitdepth_from_max(bitdepth_max) - 8; |
243 | 5.06M | int partial_sum_hv[2][8] = { { 0 } }; |
244 | 5.06M | int partial_sum_diag[2][15] = { { 0 } }; |
245 | 5.06M | int partial_sum_alt[4][11] = { { 0 } }; |
246 | | |
247 | 45.1M | for (int y = 0; y < 8; y++) { |
248 | 360M | for (int x = 0; x < 8; x++) { |
249 | 320M | const int px = (img[x] >> bitdepth_min_8) - 128; |
250 | | |
251 | 320M | partial_sum_diag[0][ y + x ] += px; |
252 | 320M | partial_sum_alt [0][ y + (x >> 1)] += px; |
253 | 320M | partial_sum_hv [0][ y ] += px; |
254 | 320M | partial_sum_alt [1][3 + y - (x >> 1)] += px; |
255 | 320M | partial_sum_diag[1][7 + y - x ] += px; |
256 | 320M | partial_sum_alt [2][3 - (y >> 1) + x ] += px; |
257 | 320M | partial_sum_hv [1][ x ] += px; |
258 | 320M | partial_sum_alt [3][ (y >> 1) + x ] += px; |
259 | 320M | } |
260 | 40.0M | img += PXSTRIDE(stride); |
261 | 40.0M | } |
262 | | |
263 | 5.06M | unsigned cost[8] = { 0 }; |
264 | 45.7M | for (int n = 0; n < 8; n++) { |
265 | 40.6M | cost[2] += partial_sum_hv[0][n] * partial_sum_hv[0][n]; |
266 | 40.6M | cost[6] += partial_sum_hv[1][n] * partial_sum_hv[1][n]; |
267 | 40.6M | } |
268 | 5.06M | cost[2] *= 105; |
269 | 5.06M | cost[6] *= 105; |
270 | | |
271 | 5.06M | static const uint16_t div_table[7] = { 840, 420, 280, 210, 168, 140, 120 }; |
272 | 40.6M | for (int n = 0; n < 7; n++) { |
273 | 35.5M | const int d = div_table[n]; |
274 | 35.5M | cost[0] += (partial_sum_diag[0][n] * partial_sum_diag[0][n] + |
275 | 35.5M | partial_sum_diag[0][14 - n] * partial_sum_diag[0][14 - n]) * d; |
276 | 35.5M | cost[4] += (partial_sum_diag[1][n] * partial_sum_diag[1][n] + |
277 | 35.5M | partial_sum_diag[1][14 - n] * partial_sum_diag[1][14 - n]) * d; |
278 | 35.5M | } |
279 | 5.06M | cost[0] += partial_sum_diag[0][7] * partial_sum_diag[0][7] * 105; |
280 | 5.06M | cost[4] += partial_sum_diag[1][7] * partial_sum_diag[1][7] * 105; |
281 | | |
282 | 25.3M | for (int n = 0; n < 4; n++) { |
283 | 20.3M | unsigned *const cost_ptr = &cost[n * 2 + 1]; |
284 | 121M | for (int m = 0; m < 5; m++) |
285 | 101M | *cost_ptr += partial_sum_alt[n][3 + m] * partial_sum_alt[n][3 + m]; |
286 | 20.3M | *cost_ptr *= 105; |
287 | 81.3M | for (int m = 0; m < 3; m++) { |
288 | 60.9M | const int d = div_table[2 * m + 1]; |
289 | 60.9M | *cost_ptr += (partial_sum_alt[n][m] * partial_sum_alt[n][m] + |
290 | 60.9M | partial_sum_alt[n][10 - m] * partial_sum_alt[n][10 - m]) * d; |
291 | 60.9M | } |
292 | 20.3M | } |
293 | | |
294 | 5.06M | int best_dir = 0; |
295 | 5.06M | unsigned best_cost = cost[0]; |
296 | 40.6M | for (int n = 1; n < 8; n++) { |
297 | 35.5M | if (cost[n] > best_cost) { |
298 | 4.99M | best_cost = cost[n]; |
299 | 4.99M | best_dir = n; |
300 | 4.99M | } |
301 | 35.5M | } |
302 | | |
303 | 5.06M | *var = (best_cost - (cost[best_dir ^ 4])) >> 10; |
304 | 5.06M | return best_dir; |
305 | 5.06M | } |
306 | | |
307 | | #if HAVE_ASM |
308 | | #if ARCH_AARCH64 || ARCH_ARM |
309 | | #include "src/arm/cdef.h" |
310 | | #elif ARCH_PPC64LE |
311 | | #include "src/ppc/cdef.h" |
312 | | #elif ARCH_RISCV |
313 | | #include "src/riscv/cdef.h" |
314 | | #elif ARCH_X86 |
315 | | #include "src/x86/cdef.h" |
316 | | #elif ARCH_LOONGARCH64 |
317 | | #include "src/loongarch/cdef.h" |
318 | | #endif |
319 | | #endif |
320 | | |
321 | 72.1k | COLD void bitfn(dav1d_cdef_dsp_init)(Dav1dCdefDSPContext *const c) { |
322 | 72.1k | c->dir = cdef_find_dir_c; |
323 | 72.1k | c->fb[0] = cdef_filter_block_8x8_c; |
324 | 72.1k | c->fb[1] = cdef_filter_block_4x8_c; |
325 | 72.1k | c->fb[2] = cdef_filter_block_4x4_c; |
326 | | |
327 | | #if HAVE_ASM |
328 | | #if ARCH_AARCH64 || ARCH_ARM |
329 | | cdef_dsp_init_arm(c); |
330 | | #elif ARCH_PPC64LE |
331 | | cdef_dsp_init_ppc(c); |
332 | | #elif ARCH_RISCV |
333 | | cdef_dsp_init_riscv(c); |
334 | | #elif ARCH_X86 |
335 | | cdef_dsp_init_x86(c); |
336 | | #elif ARCH_LOONGARCH64 |
337 | | cdef_dsp_init_loongarch(c); |
338 | | #endif |
339 | | #endif |
340 | 72.1k | } Line | Count | Source | 321 | 32.3k | COLD void bitfn(dav1d_cdef_dsp_init)(Dav1dCdefDSPContext *const c) { | 322 | 32.3k | c->dir = cdef_find_dir_c; | 323 | 32.3k | c->fb[0] = cdef_filter_block_8x8_c; | 324 | 32.3k | c->fb[1] = cdef_filter_block_4x8_c; | 325 | 32.3k | c->fb[2] = cdef_filter_block_4x4_c; | 326 | | | 327 | | #if HAVE_ASM | 328 | | #if ARCH_AARCH64 || ARCH_ARM | 329 | | cdef_dsp_init_arm(c); | 330 | | #elif ARCH_PPC64LE | 331 | | cdef_dsp_init_ppc(c); | 332 | | #elif ARCH_RISCV | 333 | | cdef_dsp_init_riscv(c); | 334 | | #elif ARCH_X86 | 335 | | cdef_dsp_init_x86(c); | 336 | | #elif ARCH_LOONGARCH64 | 337 | | cdef_dsp_init_loongarch(c); | 338 | | #endif | 339 | | #endif | 340 | 32.3k | } |
dav1d_cdef_dsp_init_16bpc Line | Count | Source | 321 | 39.7k | COLD void bitfn(dav1d_cdef_dsp_init)(Dav1dCdefDSPContext *const c) { | 322 | 39.7k | c->dir = cdef_find_dir_c; | 323 | 39.7k | c->fb[0] = cdef_filter_block_8x8_c; | 324 | 39.7k | c->fb[1] = cdef_filter_block_4x8_c; | 325 | 39.7k | c->fb[2] = cdef_filter_block_4x4_c; | 326 | | | 327 | | #if HAVE_ASM | 328 | | #if ARCH_AARCH64 || ARCH_ARM | 329 | | cdef_dsp_init_arm(c); | 330 | | #elif ARCH_PPC64LE | 331 | | cdef_dsp_init_ppc(c); | 332 | | #elif ARCH_RISCV | 333 | | cdef_dsp_init_riscv(c); | 334 | | #elif ARCH_X86 | 335 | | cdef_dsp_init_x86(c); | 336 | | #elif ARCH_LOONGARCH64 | 337 | | cdef_dsp_init_loongarch(c); | 338 | | #endif | 339 | | #endif | 340 | 39.7k | } |
|