Coverage Report

Created: 2026-08-13 07:23

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/dav1d/src/looprestoration_tmpl.c
Line
Count
Source
1
/*
2
 * Copyright © 2018, VideoLAN and dav1d authors
3
 * Copyright © 2018, Two Orioles, LLC
4
 * All rights reserved.
5
 *
6
 * Redistribution and use in source and binary forms, with or without
7
 * modification, are permitted provided that the following conditions are met:
8
 *
9
 * 1. Redistributions of source code must retain the above copyright notice, this
10
 *    list of conditions and the following disclaimer.
11
 *
12
 * 2. Redistributions in binary form must reproduce the above copyright notice,
13
 *    this list of conditions and the following disclaimer in the documentation
14
 *    and/or other materials provided with the distribution.
15
 *
16
 * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
17
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
18
 * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
19
 * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
20
 * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
21
 * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
22
 * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
23
 * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
24
 * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
25
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
26
 */
27
28
#include "config.h"
29
30
#include <stdint.h>
31
#include <stdlib.h>
32
#include <string.h>
33
34
#include "common/attributes.h"
35
#include "common/bitdepth.h"
36
#include "common/intops.h"
37
38
#include "src/looprestoration.h"
39
#include "src/tables.h"
40
41
// 256 * 1.5 + 3 + 3 = 390
42
4.57M
#define REST_UNIT_STRIDE (390)
43
44
static void wiener_filter_h(uint16_t *dst, const pixel (*left)[4],
45
                            const pixel *src, const int16_t fh[8],
46
                            const int w, const enum LrEdgeFlags edges
47
                            HIGHBD_DECL_SUFFIX)
48
4.19M
{
49
4.19M
    const int bitdepth = bitdepth_from_max(bitdepth_max);
50
4.19M
    const int round_bits_h = 3 + (bitdepth == 12) * 2;
51
4.19M
    const int rounding_off_h = 1 << (round_bits_h - 1);
52
4.19M
    const int clip_limit = 1 << (bitdepth + 1 + 7 - round_bits_h);
53
54
4.19M
    if (w < 6) {
55
        // For small widths, do the fully conditional loop with
56
        // conditions on each access.
57
2.67M
        for (int x = 0; x < w; x++) {
58
1.88M
            int sum = (1 << (bitdepth + 6));
59
1.88M
#if BITDEPTH == 8
60
1.88M
            sum += src[x] * 128;
61
1.88M
#endif
62
14.4M
            for (int i = 0; i < 7; i++) {
63
12.5M
                int idx = x + i - 3;
64
12.5M
                if (idx < 0) {
65
3.52M
                    if (!(edges & LR_HAVE_LEFT))
66
3.62M
                        sum += src[0] * fh[i];
67
18.4E
                    else if (left)
68
0
                        sum += left[0][4 + idx] * fh[i];
69
18.4E
                    else
70
18.4E
                        sum += src[idx] * fh[i];
71
9.02M
                } else if (idx >= w && !(edges & LR_HAVE_RIGHT)) {
72
3.35M
                    sum += src[w - 1] * fh[i];
73
3.35M
                } else
74
5.67M
                    sum += src[idx] * fh[i];
75
12.5M
            }
76
1.88M
            sum = iclip((sum + rounding_off_h) >> round_bits_h, 0, clip_limit - 1);
77
1.88M
            dst[x] = sum;
78
1.88M
        }
79
80
790k
        return;
81
790k
    }
82
83
    // For larger widths, do separate loops with less conditions; first
84
    // handle the start of the row.
85
3.40M
    int start = 3;
86
3.40M
    if (!(edges & LR_HAVE_LEFT)) {
87
        // If there's no left edge, pad using the leftmost pixel.
88
5.86M
        for (int x = 0; x < 3; x++) {
89
4.37M
            int sum = (1 << (bitdepth + 6));
90
4.37M
#if BITDEPTH == 8
91
4.37M
            sum += src[x] * 128;
92
4.37M
#endif
93
34.9M
            for (int i = 0; i < 7; i++) {
94
30.5M
                int idx = x + i - 3;
95
30.5M
                if (idx < 0)
96
8.74M
                    sum += src[0] * fh[i];
97
21.7M
                else
98
21.7M
                    sum += src[idx] * fh[i];
99
30.5M
            }
100
4.37M
            sum = iclip((sum + rounding_off_h) >> round_bits_h, 0, clip_limit - 1);
101
4.37M
            dst[x] = sum;
102
4.37M
        }
103
1.92M
    } else if (left) {
104
        // If we have the left edge and a separate left buffer, pad using that.
105
7.21M
        for (int x = 0; x < 3; x++) {
106
5.40M
            int sum = (1 << (bitdepth + 6));
107
5.40M
#if BITDEPTH == 8
108
5.40M
            sum += src[x] * 128;
109
5.40M
#endif
110
43.1M
            for (int i = 0; i < 7; i++) {
111
37.7M
                int idx = x + i - 3;
112
37.7M
                if (idx < 0)
113
10.8M
                    sum += left[0][4 + idx] * fh[i];
114
26.9M
                else
115
26.9M
                    sum += src[idx] * fh[i];
116
37.7M
            }
117
5.40M
            sum = iclip((sum + rounding_off_h) >> round_bits_h, 0, clip_limit - 1);
118
5.40M
            dst[x] = sum;
119
5.40M
        }
120
1.80M
    } else {
121
        // If we have the left edge, but no separate left buffer, we're in the
122
        // top/bottom area (lpf) with the left edge existing in the same
123
        // buffer; just do the regular loop from the start.
124
116k
        start = 0;
125
116k
    }
126
3.40M
    int end = w - 3;
127
3.40M
    if (edges & LR_HAVE_RIGHT)
128
1.86M
        end = w;
129
130
    // Do a condititon free loop for the bulk of the row.
131
293M
    for (int x = start; x < end; x++) {
132
289M
        int sum = (1 << (bitdepth + 6));
133
289M
#if BITDEPTH == 8
134
289M
        sum += src[x] * 128;
135
289M
#endif
136
2.31G
        for (int i = 0; i < 7; i++) {
137
2.02G
            int idx = x + i - 3;
138
2.02G
            sum += src[idx] * fh[i];
139
2.02G
        }
140
289M
        sum = iclip((sum + rounding_off_h) >> round_bits_h, 0, clip_limit - 1);
141
289M
        dst[x] = sum;
142
289M
    }
143
144
    // If we need to, calculate the end of the row with a condition for
145
    // right edge padding.
146
7.93M
    for (int x = end; x < w; x++) {
147
4.52M
        int sum = (1 << (bitdepth + 6));
148
4.52M
#if BITDEPTH == 8
149
4.52M
        sum += src[x] * 128;
150
4.52M
#endif
151
35.8M
        for (int i = 0; i < 7; i++) {
152
31.3M
            int idx = x + i - 3;
153
31.3M
            if (idx >= w)
154
8.95M
                sum += src[w - 1] * fh[i];
155
22.4M
            else
156
22.4M
                sum += src[idx] * fh[i];
157
31.3M
        }
158
4.52M
        sum = iclip((sum + rounding_off_h) >> round_bits_h, 0, clip_limit - 1);
159
4.52M
        dst[x] = sum;
160
4.52M
    }
161
3.40M
}
162
163
static void wiener_filter_v(pixel *p, uint16_t **ptrs, const int16_t fv[8],
164
                            const int w HIGHBD_DECL_SUFFIX)
165
182k
{
166
182k
    const int bitdepth = bitdepth_from_max(bitdepth_max);
167
168
182k
    const int round_bits_v = 11 - (bitdepth == 12) * 2;
169
182k
    const int rounding_off_v = 1 << (round_bits_v - 1);
170
182k
    const int round_offset = 1 << (bitdepth + (round_bits_v - 1));
171
172
10.4M
    for (int i = 0; i < w; i++) {
173
10.3M
        int sum = -round_offset;
174
175
        // Only filter using 6 input rows. The 7th row is assumed to be
176
        // identical to the last one.
177
        //
178
        // This function is assumed to only be called at the end, when doing
179
        // padding at the bottom.
180
72.0M
        for (int k = 0; k < 6; k++)
181
61.7M
            sum += ptrs[k][i] * fv[k];
182
10.3M
        sum += ptrs[5][i] * fv[6];
183
184
10.3M
        p[i] = iclip_pixel((sum + rounding_off_v) >> round_bits_v);
185
10.3M
    }
186
187
    // Shift the pointers, but only update the first 5; the 6th pointer is kept
188
    // as it was before (and the 7th is implicitly identical to the 6th).
189
1.09M
    for (int i = 0; i < 5; i++)
190
914k
        ptrs[i] = ptrs[i + 1];
191
182k
}
192
193
static void wiener_filter_hv(pixel *p, uint16_t **ptrs, const pixel (*left)[4],
194
                             const pixel *src, const int16_t filter[2][8],
195
                             const int w, const enum LrEdgeFlags edges
196
                             HIGHBD_DECL_SUFFIX)
197
3.74M
{
198
3.74M
    const int bitdepth = bitdepth_from_max(bitdepth_max);
199
200
3.74M
    const int round_bits_v = 11 - (bitdepth == 12) * 2;
201
3.74M
    const int rounding_off_v = 1 << (round_bits_v - 1);
202
3.74M
    const int round_offset = 1 << (bitdepth + (round_bits_v - 1));
203
204
3.74M
    const int16_t *fh = filter[0];
205
3.74M
    const int16_t *fv = filter[1];
206
207
    // Do combined horziontal and vertical filtering; doing horizontal
208
    // filtering of one row, combined with vertical filtering of 6
209
    // preexisting rows and the newly filtered row.
210
211
    // For simplicity in the C implementation, just do a separate call
212
    // of the horizontal filter, into a temporary buffer.
213
3.74M
    uint16_t tmp[REST_UNIT_STRIDE];
214
3.74M
    wiener_filter_h(tmp, left, src, fh, w, edges HIGHBD_TAIL_SUFFIX);
215
216
280M
    for (int i = 0; i < w; i++) {
217
276M
        int sum = -round_offset;
218
219
        // Filter using the 6 stored preexisting rows, and the newly
220
        // filtered one in tmp[].
221
1.92G
        for (int k = 0; k < 6; k++)
222
1.65G
            sum += ptrs[k][i] * fv[k];
223
276M
        sum += tmp[i] * fv[6];
224
        // At this point, after having read all inputs at point [i], we
225
        // could overwrite [i] with the newly filtered data.
226
227
276M
        p[i] = iclip_pixel((sum + rounding_off_v) >> round_bits_v);
228
276M
    }
229
230
    // For simplicity in the C implementation, just memcpy the newly
231
    // filtered row into ptrs[6]. Normally, in steady state filtering,
232
    // this output row, ptrs[6], is equal to ptrs[0]. However at startup,
233
    // at the top of the filtered area, we may have ptrs[0] equal to ptrs[1],
234
    // so we can't assume we can write into ptrs[0] but we need to keep
235
    // a separate pointer for the next row to write into.
236
3.74M
    memcpy(ptrs[6], tmp, sizeof(uint16_t) * REST_UNIT_STRIDE);
237
238
    // Rotate the window of pointers. Shift the 6 pointers downwards one step.
239
26.0M
    for (int i = 0; i < 6; i++)
240
22.2M
        ptrs[i] = ptrs[i + 1];
241
    // The topmost pointer, ptrs[6], which isn't used as input, is set to
242
    // ptrs[0], which will be used as output for the next _hv call.
243
    // At the start of the filtering, the caller may set ptrs[6] to the
244
    // right next buffer to fill in, instead.
245
3.74M
    ptrs[6] = ptrs[0];
246
3.74M
}
247
248
// FIXME Could split into luma and chroma specific functions,
249
// (since first and last tops are always 0 for chroma)
250
static void wiener_c(pixel *p, const ptrdiff_t stride,
251
                     const pixel (*left)[4],
252
                     const pixel *lpf, const int w, int h,
253
                     const LooprestorationParams *const params,
254
                     const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
255
122k
{
256
    // Values stored between horizontal and vertical filtering don't
257
    // fit in a uint8_t.
258
122k
    uint16_t hor[6 * REST_UNIT_STRIDE];
259
122k
    uint16_t *ptrs[7], *rows[6];
260
860k
    for (int i = 0; i < 6; i++)
261
737k
        rows[i] = &hor[i * REST_UNIT_STRIDE];
262
122k
    const int16_t (*const filter)[8] = params->filter;
263
122k
    const int16_t *fh = params->filter[0];
264
122k
    const int16_t *fv = params->filter[1];
265
122k
    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
266
267
122k
    const pixel *src = p;
268
122k
    if (edges & LR_HAVE_TOP) {
269
70.2k
        ptrs[0] = rows[0];
270
70.2k
        ptrs[1] = rows[0];
271
70.2k
        ptrs[2] = rows[1];
272
70.2k
        ptrs[3] = rows[2];
273
70.2k
        ptrs[4] = rows[2];
274
70.2k
        ptrs[5] = rows[2];
275
276
70.2k
        wiener_filter_h(rows[0], NULL, lpf, fh, w, edges HIGHBD_TAIL_SUFFIX);
277
70.2k
        lpf += PXSTRIDE(stride);
278
70.2k
        wiener_filter_h(rows[1], NULL, lpf, fh, w, edges HIGHBD_TAIL_SUFFIX);
279
280
70.2k
        wiener_filter_h(rows[2], left, src, fh, w, edges HIGHBD_TAIL_SUFFIX);
281
70.2k
        left++;
282
70.2k
        src += PXSTRIDE(stride);
283
284
70.2k
        if (--h <= 0)
285
553
            goto v1;
286
287
69.7k
        ptrs[4] = ptrs[5] = rows[3];
288
69.7k
        wiener_filter_h(rows[3], left, src, fh, w, edges HIGHBD_TAIL_SUFFIX);
289
69.7k
        left++;
290
69.7k
        src += PXSTRIDE(stride);
291
292
69.7k
        if (--h <= 0)
293
887
            goto v2;
294
295
68.8k
        ptrs[5] = rows[4];
296
68.8k
        wiener_filter_h(rows[4], left, src, fh, w, edges HIGHBD_TAIL_SUFFIX);
297
68.8k
        left++;
298
68.8k
        src += PXSTRIDE(stride);
299
300
68.8k
        if (--h <= 0)
301
614
            goto v3;
302
68.8k
    } else {
303
52.7k
        ptrs[0] = rows[0];
304
52.7k
        ptrs[1] = rows[0];
305
52.7k
        ptrs[2] = rows[0];
306
52.7k
        ptrs[3] = rows[0];
307
52.7k
        ptrs[4] = rows[0];
308
52.7k
        ptrs[5] = rows[0];
309
310
52.7k
        wiener_filter_h(rows[0], left, src, fh, w, edges HIGHBD_TAIL_SUFFIX);
311
52.7k
        left++;
312
52.7k
        src += PXSTRIDE(stride);
313
314
52.7k
        if (--h <= 0)
315
16.3k
            goto v1;
316
317
36.3k
        ptrs[4] = ptrs[5] = rows[1];
318
36.3k
        wiener_filter_h(rows[1], left, src, fh, w, edges HIGHBD_TAIL_SUFFIX);
319
36.3k
        left++;
320
36.3k
        src += PXSTRIDE(stride);
321
322
36.3k
        if (--h <= 0)
323
9.71k
            goto v2;
324
325
26.6k
        ptrs[5] = rows[2];
326
26.6k
        wiener_filter_h(rows[2], left, src, fh, w, edges HIGHBD_TAIL_SUFFIX);
327
26.6k
        left++;
328
26.6k
        src += PXSTRIDE(stride);
329
330
26.6k
        if (--h <= 0)
331
1.27k
            goto v3;
332
333
25.3k
        ptrs[6] = rows[3];
334
25.3k
        wiener_filter_hv(p, ptrs, left, src, filter, w, edges
335
25.3k
                         HIGHBD_TAIL_SUFFIX);
336
25.3k
        left++;
337
25.3k
        src += PXSTRIDE(stride);
338
25.3k
        p += PXSTRIDE(stride);
339
340
25.3k
        if (--h <= 0)
341
1.56k
            goto v3;
342
343
23.7k
        ptrs[6] = rows[4];
344
23.7k
        wiener_filter_hv(p, ptrs, left, src, filter, w, edges
345
23.7k
                         HIGHBD_TAIL_SUFFIX);
346
23.7k
        left++;
347
23.7k
        src += PXSTRIDE(stride);
348
23.7k
        p += PXSTRIDE(stride);
349
350
23.7k
        if (--h <= 0)
351
1.25k
            goto v3;
352
23.7k
    }
353
354
90.7k
    ptrs[6] = ptrs[5] + REST_UNIT_STRIDE;
355
3.56M
    do {
356
3.56M
        wiener_filter_hv(p, ptrs, left, src, filter, w, edges
357
3.56M
                         HIGHBD_TAIL_SUFFIX);
358
3.56M
        left++;
359
3.56M
        src += PXSTRIDE(stride);
360
3.56M
        p += PXSTRIDE(stride);
361
3.56M
    } while (--h > 0);
362
363
90.7k
    if (!(edges & LR_HAVE_BOTTOM))
364
19.9k
        goto v3;
365
366
70.8k
    wiener_filter_hv(p, ptrs, NULL, lpf_bottom, filter, w, edges
367
70.8k
                     HIGHBD_TAIL_SUFFIX);
368
70.8k
    lpf_bottom += PXSTRIDE(stride);
369
70.8k
    p += PXSTRIDE(stride);
370
371
70.8k
    wiener_filter_hv(p, ptrs, NULL, lpf_bottom, filter, w, edges
372
70.8k
                     HIGHBD_TAIL_SUFFIX);
373
70.8k
    p += PXSTRIDE(stride);
374
122k
v1:
375
122k
    wiener_filter_v(p, ptrs, fv, w HIGHBD_TAIL_SUFFIX);
376
377
122k
    return;
378
379
24.6k
v3:
380
24.6k
    wiener_filter_v(p, ptrs, fv, w HIGHBD_TAIL_SUFFIX);
381
24.6k
    p += PXSTRIDE(stride);
382
35.2k
v2:
383
35.2k
    wiener_filter_v(p, ptrs, fv, w HIGHBD_TAIL_SUFFIX);
384
35.2k
    p += PXSTRIDE(stride);
385
35.2k
    goto v1;
386
24.6k
}
387
388
// SGR
389
static NOINLINE void rotate(int32_t **sumsq_ptrs, coef **sum_ptrs, int n)
390
7.74M
{
391
7.74M
    int32_t *tmp32 = sumsq_ptrs[0];
392
7.74M
    coef *tmpc = sum_ptrs[0];
393
23.7M
    for (int i = 0; i < n - 1; i++) {
394
16.0M
        sumsq_ptrs[i] = sumsq_ptrs[i + 1];
395
16.0M
        sum_ptrs[i] = sum_ptrs[i + 1];
396
16.0M
    }
397
7.74M
    sumsq_ptrs[n - 1] = tmp32;
398
7.74M
    sum_ptrs[n - 1] = tmpc;
399
7.74M
}
400
401
static NOINLINE void rotate5_x2(int32_t **sumsq_ptrs, coef **sum_ptrs)
402
1.69M
{
403
1.69M
    int32_t *tmp32[2];
404
1.69M
    coef *tmpc[2];
405
5.07M
    for (int i = 0; i < 2; i++) {
406
3.38M
        tmp32[i] = sumsq_ptrs[i];
407
3.38M
        tmpc[i] = sum_ptrs[i];
408
3.38M
    }
409
6.75M
    for (int i = 0; i < 3; i++) {
410
5.06M
        sumsq_ptrs[i] = sumsq_ptrs[i + 2];
411
5.06M
        sum_ptrs[i] = sum_ptrs[i + 2];
412
5.06M
    }
413
5.07M
    for (int i = 0; i < 2; i++) {
414
3.38M
        sumsq_ptrs[3 + i] = tmp32[i];
415
3.38M
        sum_ptrs[3 + i] = tmpc[i];
416
3.38M
    }
417
1.69M
}
418
419
static NOINLINE void sgr_box3_row_h(int32_t *sumsq, coef *sum,
420
                                    const pixel (*left)[4],
421
                                    const pixel *src, const int w,
422
                                    const enum LrEdgeFlags edges)
423
3.09M
{
424
3.09M
    sumsq++;
425
3.09M
    sum++;
426
3.09M
    int a = edges & LR_HAVE_LEFT ? (left ? left[0][2] : src[-2]) : src[0];
427
3.09M
    int b = edges & LR_HAVE_LEFT ? (left ? left[0][3] : src[-1]) : src[0];
428
189M
    for (int x = -1; x < w + 1; x++) {
429
186M
        int c = (x + 1 < w || (edges & LR_HAVE_RIGHT)) ? src[x + 1] : src[w - 1];
430
186M
        sum[x] = a + b + c;
431
186M
        sumsq[x] = a * a + b * b + c * c;
432
186M
        a = b;
433
186M
        b = c;
434
186M
    }
435
3.09M
}
436
437
static NOINLINE void sgr_box5_row_h(int32_t *sumsq, coef *sum,
438
                                    const pixel (*left)[4],
439
                                    const pixel *src, const int w,
440
                                    const enum LrEdgeFlags edges)
441
3.40M
{
442
3.40M
    sumsq++;
443
3.40M
    sum++;
444
3.40M
    int a = edges & LR_HAVE_LEFT ? (left ? left[0][1] : src[-3]) : src[0];
445
3.40M
    int b = edges & LR_HAVE_LEFT ? (left ? left[0][2] : src[-2]) : src[0];
446
3.40M
    int c = edges & LR_HAVE_LEFT ? (left ? left[0][3] : src[-1]) : src[0];
447
3.40M
    int d = src[0];
448
201M
    for (int x = -1; x < w + 1; x++) {
449
197M
        int e = (x + 2 < w || (edges & LR_HAVE_RIGHT)) ? src[x + 2] : src[w - 1];
450
197M
        sum[x] = a + b + c + d + e;
451
197M
        sumsq[x] = a * a + b * b + c * c + d * d + e * e;
452
197M
        a = b;
453
197M
        b = c;
454
197M
        c = d;
455
197M
        d = e;
456
197M
    }
457
3.40M
}
458
459
static void sgr_box35_row_h(int32_t *sumsq3, coef *sum3,
460
                            int32_t *sumsq5, coef *sum5,
461
                            const pixel (*left)[4],
462
                            const pixel *src, const int w,
463
                            const enum LrEdgeFlags edges)
464
2.28M
{
465
2.28M
    sgr_box3_row_h(sumsq3, sum3, left, src, w, edges);
466
2.28M
    sgr_box5_row_h(sumsq5, sum5, left, src, w, edges);
467
2.28M
}
468
469
static NOINLINE void sgr_box3_row_v(int32_t **sumsq, coef **sum,
470
                                    int32_t *sumsq_out, coef *sum_out,
471
                                    const int w)
472
3.05M
{
473
186M
    for (int x = 0; x < w + 2; x++) {
474
183M
        int sq_a = sumsq[0][x];
475
183M
        int sq_b = sumsq[1][x];
476
183M
        int sq_c = sumsq[2][x];
477
183M
        int s_a = sum[0][x];
478
183M
        int s_b = sum[1][x];
479
183M
        int s_c = sum[2][x];
480
183M
        sumsq_out[x] = sq_a + sq_b + sq_c;
481
183M
        sum_out[x] = s_a + s_b + s_c;
482
183M
    }
483
3.05M
}
484
485
static NOINLINE void sgr_box5_row_v(int32_t **sumsq, coef **sum,
486
                                    int32_t *sumsq_out, coef *sum_out,
487
                                    const int w)
488
1.69M
{
489
100M
    for (int x = 0; x < w + 2; x++) {
490
98.6M
        int sq_a = sumsq[0][x];
491
98.6M
        int sq_b = sumsq[1][x];
492
98.6M
        int sq_c = sumsq[2][x];
493
98.6M
        int sq_d = sumsq[3][x];
494
98.6M
        int sq_e = sumsq[4][x];
495
98.6M
        int s_a = sum[0][x];
496
98.6M
        int s_b = sum[1][x];
497
98.6M
        int s_c = sum[2][x];
498
98.6M
        int s_d = sum[3][x];
499
98.6M
        int s_e = sum[4][x];
500
98.6M
        sumsq_out[x] = sq_a + sq_b + sq_c + sq_d + sq_e;
501
98.6M
        sum_out[x] = s_a + s_b + s_c + s_d + s_e;
502
98.6M
    }
503
1.69M
}
504
505
static NOINLINE void sgr_calc_row_ab(int32_t *AA, coef *BB, int w, int s,
506
                                     int bitdepth_max, int n, int sgr_one_by_x)
507
4.73M
{
508
4.73M
    const int bitdepth_min_8 = bitdepth_from_max(bitdepth_max) - 8;
509
273M
    for (int i = 0; i < w + 2; i++) {
510
268M
        const int a =
511
268M
            (AA[i] + ((1 << (2 * bitdepth_min_8)) >> 1)) >> (2 * bitdepth_min_8);
512
268M
        const int b =
513
268M
            (BB[i] + ((1 << bitdepth_min_8) >> 1)) >> bitdepth_min_8;
514
515
268M
        const unsigned p = imax(a * n - b * b, 0);
516
268M
        const unsigned z = (p * s + (1 << 19)) >> 20;
517
268M
        const unsigned x = dav1d_sgr_x_by_x[umin(z, 255)];
518
519
        // This is where we invert A and B, so that B is of size coef.
520
268M
        AA[i] = (x * BB[i] * sgr_one_by_x + (1 << 11)) >> 12;
521
268M
        BB[i] = x;
522
268M
    }
523
4.73M
}
524
525
static void sgr_box3_vert(int32_t **sumsq, coef **sum,
526
                          int32_t *sumsq_out, coef *sum_out,
527
                          const int w, const int s, const int bitdepth_max)
528
3.05M
{
529
3.05M
    sgr_box3_row_v(sumsq, sum, sumsq_out, sum_out, w);
530
3.05M
    sgr_calc_row_ab(sumsq_out, sum_out, w, s, bitdepth_max, 9, 455);
531
3.05M
    rotate(sumsq, sum, 3);
532
3.05M
}
533
534
static void sgr_box5_vert(int32_t **sumsq, coef **sum,
535
                          int32_t *sumsq_out, coef *sum_out,
536
                          const int w, const int s, const int bitdepth_max)
537
1.69M
{
538
1.69M
    sgr_box5_row_v(sumsq, sum, sumsq_out, sum_out, w);
539
1.69M
    sgr_calc_row_ab(sumsq_out, sum_out, w, s, bitdepth_max, 25, 164);
540
1.69M
    rotate5_x2(sumsq, sum);
541
1.69M
}
542
543
static void sgr_box3_hv(int32_t **sumsq, coef **sum,
544
                        int32_t *AA, coef *BB,
545
                        const pixel (*left)[4],
546
                        const pixel *src, const int w,
547
                        const int s,
548
                        const enum LrEdgeFlags edges,
549
                        const int bitdepth_max)
550
784k
{
551
784k
    sgr_box3_row_h(sumsq[2], sum[2], left, src, w, edges);
552
784k
    sgr_box3_vert(sumsq, sum, AA, BB, w, s, bitdepth_max);
553
784k
}
554
555
static NOINLINE void sgr_finish_filter_row1(coef *tmp,
556
                                            const pixel *src,
557
                                            int32_t **A_ptrs, coef **B_ptrs,
558
                                            const int w)
559
2.90M
{
560
2.90M
#define EIGHT_NEIGHBORS(P, i)\
561
333M
    ((P[1][i] + P[1][i - 1] + P[1][i + 1] + P[0][i] + P[2][i]) * 4 + \
562
333M
     (P[0][i - 1] + P[2][i - 1] +                           \
563
333M
      P[0][i + 1] + P[2][i + 1]) * 3)
564
169M
    for (int i = 0; i < w; i++) {
565
166M
        const int a = EIGHT_NEIGHBORS(B_ptrs, i + 1);
566
166M
        const int b = EIGHT_NEIGHBORS(A_ptrs, i + 1);
567
166M
        tmp[i] = (b - a * src[i] + (1 << 8)) >> 9;
568
166M
    }
569
2.90M
#undef EIGHT_NEIGHBORS
570
2.90M
}
571
572
7.47M
#define FILTER_OUT_STRIDE (384)
573
574
static NOINLINE void sgr_finish_filter2(coef *tmp,
575
                                        const pixel *src,
576
                                        const ptrdiff_t src_stride,
577
                                        int32_t **A_ptrs, coef **B_ptrs,
578
                                        const int w, const int h)
579
1.61M
{
580
1.61M
#define SIX_NEIGHBORS(P, i)\
581
180M
    ((P[0][i]     + P[1][i]) * 6 +   \
582
180M
     (P[0][i - 1] + P[1][i - 1] +    \
583
180M
      P[0][i + 1] + P[1][i + 1]) * 5)
584
91.6M
    for (int i = 0; i < w; i++) {
585
90.0M
        const int a = SIX_NEIGHBORS(B_ptrs, i + 1);
586
90.0M
        const int b = SIX_NEIGHBORS(A_ptrs, i + 1);
587
90.0M
        tmp[i] = (b - a * src[i] + (1 << 8)) >> 9;
588
90.0M
    }
589
1.61M
    if (h <= 1)
590
18.9k
        return;
591
1.59M
    tmp += FILTER_OUT_STRIDE;
592
1.59M
    src += PXSTRIDE(src_stride);
593
1.59M
    const int32_t *A = &A_ptrs[1][1];
594
1.59M
    const coef *B = &B_ptrs[1][1];
595
90.8M
    for (int i = 0; i < w; i++) {
596
89.2M
        const int a = B[i] * 6 + (B[i - 1] + B[i + 1]) * 5;
597
89.2M
        const int b = A[i] * 6 + (A[i - 1] + A[i + 1]) * 5;
598
89.2M
        tmp[i] = (b - a * src[i] + (1 << 7)) >> 8;
599
89.2M
    }
600
1.59M
#undef SIX_NEIGHBORS
601
1.59M
}
602
603
static NOINLINE void sgr_weighted_row1(pixel *dst, const coef *t1,
604
                                       const int w, const int w1 HIGHBD_DECL_SUFFIX)
605
1.81M
{
606
97.0M
    for (int i = 0; i < w; i++) {
607
95.2M
        const int v = w1 * t1[i];
608
95.2M
        dst[i] = iclip_pixel(dst[i] + ((v + (1 << 10)) >> 11));
609
95.2M
    }
610
1.81M
}
611
612
static NOINLINE void sgr_weighted2(pixel *dst, const ptrdiff_t dst_stride,
613
                                   const coef *t1, const coef *t2,
614
                                   const int w, const int h,
615
                                   const int w0, const int w1 HIGHBD_DECL_SUFFIX)
616
1.08M
{
617
3.22M
    for (int j = 0; j < h; j++) {
618
125M
        for (int i = 0; i < w; i++) {
619
123M
            const int v = w0 * t1[i] + w1 * t2[i];
620
123M
            dst[i] = iclip_pixel(dst[i] + ((v + (1 << 10)) >> 11));
621
123M
        }
622
2.14M
        dst += PXSTRIDE(dst_stride);
623
2.14M
        t1 += FILTER_OUT_STRIDE;
624
2.14M
        t2 += FILTER_OUT_STRIDE;
625
2.14M
    }
626
1.08M
}
627
628
static NOINLINE void sgr_finish1(pixel **dst, const ptrdiff_t stride,
629
                                 int32_t **A_ptrs, coef **B_ptrs, const int w,
630
                                 const int w1 HIGHBD_DECL_SUFFIX)
631
765k
{
632
    // Only one single row, no stride needed
633
765k
    ALIGN_STK_16(coef, tmp, 384,);
634
635
765k
    sgr_finish_filter_row1(tmp, *dst, A_ptrs, B_ptrs, w);
636
765k
    sgr_weighted_row1(*dst, tmp, w, w1 HIGHBD_TAIL_SUFFIX);
637
765k
    *dst += PXSTRIDE(stride);
638
765k
    rotate(A_ptrs, B_ptrs, 3);
639
765k
}
640
641
static NOINLINE void sgr_finish2(pixel **dst, const ptrdiff_t stride,
642
                                 int32_t **A_ptrs, coef **B_ptrs,
643
                                 const int w, const int h, const int w1
644
                                 HIGHBD_DECL_SUFFIX)
645
530k
{
646
530k
    ALIGN_STK_16(coef, tmp, 2*FILTER_OUT_STRIDE,);
647
648
530k
    sgr_finish_filter2(tmp, *dst, stride, A_ptrs, B_ptrs, w, h);
649
530k
    sgr_weighted_row1(*dst, tmp, w, w1 HIGHBD_TAIL_SUFFIX);
650
530k
    *dst += PXSTRIDE(stride);
651
530k
    if (h > 1) {
652
524k
        sgr_weighted_row1(*dst, tmp + FILTER_OUT_STRIDE, w, w1 HIGHBD_TAIL_SUFFIX);
653
524k
        *dst += PXSTRIDE(stride);
654
524k
    }
655
530k
    rotate(A_ptrs, B_ptrs, 2);
656
530k
}
657
658
static NOINLINE void sgr_finish_mix(pixel **dst, const ptrdiff_t stride,
659
                                    int32_t **A5_ptrs, coef **B5_ptrs,
660
                                    int32_t **A3_ptrs, coef **B3_ptrs,
661
                                    const int w, const int h,
662
                                    const int w0, const int w1 HIGHBD_DECL_SUFFIX)
663
1.08M
{
664
1.08M
    ALIGN_STK_16(coef, tmp5, 2*FILTER_OUT_STRIDE,);
665
1.08M
    ALIGN_STK_16(coef, tmp3, 2*FILTER_OUT_STRIDE,);
666
667
1.08M
    sgr_finish_filter2(tmp5, *dst, stride, A5_ptrs, B5_ptrs, w, h);
668
1.08M
    sgr_finish_filter_row1(tmp3, *dst, A3_ptrs, B3_ptrs, w);
669
1.08M
    if (h > 1)
670
1.06M
        sgr_finish_filter_row1(tmp3 + FILTER_OUT_STRIDE, *dst + PXSTRIDE(stride),
671
1.06M
                               &A3_ptrs[1], &B3_ptrs[1], w);
672
1.08M
    sgr_weighted2(*dst, stride, tmp5, tmp3, w, h, w0, w1 HIGHBD_TAIL_SUFFIX);
673
1.08M
    *dst += h*PXSTRIDE(stride);
674
1.08M
    rotate(A5_ptrs, B5_ptrs, 2);
675
1.08M
    rotate(A3_ptrs, B3_ptrs, 4);
676
1.08M
}
677
678
679
static void sgr_3x3_c(pixel *dst, const ptrdiff_t stride,
680
                      const pixel (*left)[4], const pixel *lpf,
681
                      const int w, int h,
682
                      const LooprestorationParams *const params,
683
                      const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
684
18.6k
{
685
2.15M
#define BUF_STRIDE (384 + 16)
686
18.6k
    ALIGN_STK_16(int32_t, sumsq_buf, BUF_STRIDE * 3 + 16,);
687
18.6k
    ALIGN_STK_16(coef, sum_buf, BUF_STRIDE * 3 + 16,);
688
18.6k
    int32_t *sumsq_ptrs[3], *sumsq_rows[3];
689
18.6k
    coef *sum_ptrs[3], *sum_rows[3];
690
74.6k
    for (int i = 0; i < 3; i++) {
691
55.9k
        sumsq_rows[i] = &sumsq_buf[i * BUF_STRIDE];
692
55.9k
        sum_rows[i] = &sum_buf[i * BUF_STRIDE];
693
55.9k
    }
694
695
18.6k
    ALIGN_STK_16(int32_t, A_buf, BUF_STRIDE * 3 + 16,);
696
18.6k
    ALIGN_STK_16(coef, B_buf, BUF_STRIDE * 3 + 16,);
697
18.6k
    int32_t *A_ptrs[3];
698
18.6k
    coef *B_ptrs[3];
699
74.6k
    for (int i = 0; i < 3; i++) {
700
55.9k
        A_ptrs[i] = &A_buf[i * BUF_STRIDE];
701
55.9k
        B_ptrs[i] = &B_buf[i * BUF_STRIDE];
702
55.9k
    }
703
18.6k
    const pixel *src = dst;
704
18.6k
    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
705
706
18.6k
    if (edges & LR_HAVE_TOP) {
707
13.5k
        sumsq_ptrs[0] = sumsq_rows[0];
708
13.5k
        sumsq_ptrs[1] = sumsq_rows[1];
709
13.5k
        sumsq_ptrs[2] = sumsq_rows[2];
710
13.5k
        sum_ptrs[0] = sum_rows[0];
711
13.5k
        sum_ptrs[1] = sum_rows[1];
712
13.5k
        sum_ptrs[2] = sum_rows[2];
713
714
13.5k
        sgr_box3_row_h(sumsq_rows[0], sum_rows[0], NULL, lpf, w, edges);
715
13.5k
        lpf += PXSTRIDE(stride);
716
13.5k
        sgr_box3_row_h(sumsq_rows[1], sum_rows[1], NULL, lpf, w, edges);
717
718
13.5k
        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
719
13.5k
                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
720
13.5k
        left++;
721
13.5k
        src += PXSTRIDE(stride);
722
13.5k
        rotate(A_ptrs, B_ptrs, 3);
723
724
13.5k
        if (--h <= 0)
725
372
            goto vert_1;
726
727
13.1k
        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
728
13.1k
                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
729
13.1k
        left++;
730
13.1k
        src += PXSTRIDE(stride);
731
13.1k
        rotate(A_ptrs, B_ptrs, 3);
732
733
13.1k
        if (--h <= 0)
734
288
            goto vert_2;
735
13.1k
    } else {
736
5.11k
        sumsq_ptrs[0] = sumsq_rows[0];
737
5.11k
        sumsq_ptrs[1] = sumsq_rows[0];
738
5.11k
        sumsq_ptrs[2] = sumsq_rows[0];
739
5.11k
        sum_ptrs[0] = sum_rows[0];
740
5.11k
        sum_ptrs[1] = sum_rows[0];
741
5.11k
        sum_ptrs[2] = sum_rows[0];
742
743
5.11k
        sgr_box3_row_h(sumsq_rows[0], sum_rows[0], left, src, w, edges);
744
5.11k
        left++;
745
5.11k
        src += PXSTRIDE(stride);
746
747
5.11k
        sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
748
5.11k
                      w, params->sgr.s1, BITDEPTH_MAX);
749
5.11k
        rotate(A_ptrs, B_ptrs, 3);
750
751
5.11k
        if (--h <= 0)
752
941
            goto vert_1;
753
754
4.16k
        sumsq_ptrs[2] = sumsq_rows[1];
755
4.16k
        sum_ptrs[2] = sum_rows[1];
756
757
4.16k
        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
758
4.16k
                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
759
4.16k
        left++;
760
4.16k
        src += PXSTRIDE(stride);
761
4.16k
        rotate(A_ptrs, B_ptrs, 3);
762
763
4.16k
        if (--h <= 0)
764
775
            goto vert_2;
765
766
3.39k
        sumsq_ptrs[2] = sumsq_rows[2];
767
3.39k
        sum_ptrs[2] = sum_rows[2];
768
3.39k
    }
769
770
730k
    do {
771
730k
        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
772
730k
                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
773
730k
        left++;
774
730k
        src += PXSTRIDE(stride);
775
776
730k
        sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
777
730k
                    w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
778
730k
    } while (--h > 0);
779
780
16.2k
    if (!(edges & LR_HAVE_BOTTOM))
781
4.23k
        goto vert_2;
782
783
12.0k
    sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
784
12.0k
                NULL, lpf_bottom, w, params->sgr.s1, edges, BITDEPTH_MAX);
785
12.0k
    lpf_bottom += PXSTRIDE(stride);
786
787
12.0k
    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
788
12.0k
                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
789
790
12.0k
    sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
791
12.0k
                NULL, lpf_bottom, w, params->sgr.s1, edges, BITDEPTH_MAX);
792
793
12.0k
    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
794
12.0k
                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
795
12.0k
    return;
796
797
5.30k
vert_2:
798
5.30k
    sumsq_ptrs[2] = sumsq_ptrs[1];
799
5.30k
    sum_ptrs[2] = sum_ptrs[1];
800
5.30k
    sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
801
5.30k
                  w, params->sgr.s1, BITDEPTH_MAX);
802
803
5.30k
    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
804
5.30k
                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
805
806
6.61k
output_1:
807
6.61k
    sumsq_ptrs[2] = sumsq_ptrs[1];
808
6.61k
    sum_ptrs[2] = sum_ptrs[1];
809
6.61k
    sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
810
6.61k
                  w, params->sgr.s1, BITDEPTH_MAX);
811
812
6.61k
    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
813
6.61k
                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
814
6.61k
    return;
815
816
1.31k
vert_1:
817
1.31k
    sumsq_ptrs[2] = sumsq_ptrs[1];
818
1.31k
    sum_ptrs[2] = sum_ptrs[1];
819
1.31k
    sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
820
1.31k
                  w, params->sgr.s1, BITDEPTH_MAX);
821
1.31k
    rotate(A_ptrs, B_ptrs, 3);
822
1.31k
    goto output_1;
823
5.30k
}
824
825
static void sgr_5x5_c(pixel *dst, const ptrdiff_t stride,
826
                      const pixel (*left)[4], const pixel *lpf,
827
                      const int w, int h,
828
                      const LooprestorationParams *const params,
829
                      const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
830
26.1k
{
831
26.1k
    ALIGN_STK_16(int32_t, sumsq_buf, BUF_STRIDE * 5 + 16,);
832
26.1k
    ALIGN_STK_16(coef, sum_buf, BUF_STRIDE * 5 + 16,);
833
26.1k
    int32_t *sumsq_ptrs[5], *sumsq_rows[5];
834
26.1k
    coef *sum_ptrs[5], *sum_rows[5];
835
156k
    for (int i = 0; i < 5; i++) {
836
130k
        sumsq_rows[i] = &sumsq_buf[i * BUF_STRIDE];
837
130k
        sum_rows[i] = &sum_buf[i * BUF_STRIDE];
838
130k
    }
839
840
26.1k
    ALIGN_STK_16(int32_t, A_buf, BUF_STRIDE * 2 + 16,);
841
26.1k
    ALIGN_STK_16(coef, B_buf, BUF_STRIDE * 2 + 16,);
842
26.1k
    int32_t *A_ptrs[2];
843
26.1k
    coef *B_ptrs[2];
844
78.3k
    for (int i = 0; i < 2; i++) {
845
52.2k
        A_ptrs[i] = &A_buf[i * BUF_STRIDE];
846
52.2k
        B_ptrs[i] = &B_buf[i * BUF_STRIDE];
847
52.2k
    }
848
26.1k
    const pixel *src = dst;
849
26.1k
    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
850
851
26.1k
    if (edges & LR_HAVE_TOP) {
852
17.3k
        sumsq_ptrs[0] = sumsq_rows[0];
853
17.3k
        sumsq_ptrs[1] = sumsq_rows[0];
854
17.3k
        sumsq_ptrs[2] = sumsq_rows[1];
855
17.3k
        sumsq_ptrs[3] = sumsq_rows[2];
856
17.3k
        sumsq_ptrs[4] = sumsq_rows[3];
857
17.3k
        sum_ptrs[0] = sum_rows[0];
858
17.3k
        sum_ptrs[1] = sum_rows[0];
859
17.3k
        sum_ptrs[2] = sum_rows[1];
860
17.3k
        sum_ptrs[3] = sum_rows[2];
861
17.3k
        sum_ptrs[4] = sum_rows[3];
862
863
17.3k
        sgr_box5_row_h(sumsq_rows[0], sum_rows[0], NULL, lpf, w, edges);
864
17.3k
        lpf += PXSTRIDE(stride);
865
17.3k
        sgr_box5_row_h(sumsq_rows[1], sum_rows[1], NULL, lpf, w, edges);
866
867
17.3k
        sgr_box5_row_h(sumsq_rows[2], sum_rows[2], left, src, w, edges);
868
17.3k
        left++;
869
17.3k
        src += PXSTRIDE(stride);
870
871
17.3k
        if (--h <= 0)
872
377
            goto vert_1;
873
874
16.9k
        sgr_box5_row_h(sumsq_rows[3], sum_rows[3], left, src, w, edges);
875
16.9k
        left++;
876
16.9k
        src += PXSTRIDE(stride);
877
16.9k
        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
878
16.9k
                      w, params->sgr.s0, BITDEPTH_MAX);
879
16.9k
        rotate(A_ptrs, B_ptrs, 2);
880
881
16.9k
        if (--h <= 0)
882
514
            goto vert_2;
883
884
        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
885
        // one of them to point at the previously unused rows[4].
886
16.4k
        sumsq_ptrs[3] = sumsq_rows[4];
887
16.4k
        sum_ptrs[3] = sum_rows[4];
888
16.4k
    } else {
889
8.81k
        sumsq_ptrs[0] = sumsq_rows[0];
890
8.81k
        sumsq_ptrs[1] = sumsq_rows[0];
891
8.81k
        sumsq_ptrs[2] = sumsq_rows[0];
892
8.81k
        sumsq_ptrs[3] = sumsq_rows[0];
893
8.81k
        sumsq_ptrs[4] = sumsq_rows[0];
894
8.81k
        sum_ptrs[0] = sum_rows[0];
895
8.81k
        sum_ptrs[1] = sum_rows[0];
896
8.81k
        sum_ptrs[2] = sum_rows[0];
897
8.81k
        sum_ptrs[3] = sum_rows[0];
898
8.81k
        sum_ptrs[4] = sum_rows[0];
899
900
8.81k
        sgr_box5_row_h(sumsq_rows[0], sum_rows[0], left, src, w, edges);
901
8.81k
        left++;
902
8.81k
        src += PXSTRIDE(stride);
903
904
8.81k
        if (--h <= 0)
905
1.15k
            goto vert_1;
906
907
7.66k
        sumsq_ptrs[4] = sumsq_rows[1];
908
7.66k
        sum_ptrs[4] = sum_rows[1];
909
910
7.66k
        sgr_box5_row_h(sumsq_rows[1], sum_rows[1], left, src, w, edges);
911
7.66k
        left++;
912
7.66k
        src += PXSTRIDE(stride);
913
914
7.66k
        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
915
7.66k
                      w, params->sgr.s0, BITDEPTH_MAX);
916
7.66k
        rotate(A_ptrs, B_ptrs, 2);
917
918
7.66k
        if (--h <= 0)
919
972
            goto vert_2;
920
921
6.69k
        sumsq_ptrs[3] = sumsq_rows[2];
922
6.69k
        sumsq_ptrs[4] = sumsq_rows[3];
923
6.69k
        sum_ptrs[3] = sum_rows[2];
924
6.69k
        sum_ptrs[4] = sum_rows[3];
925
926
6.69k
        sgr_box5_row_h(sumsq_rows[2], sum_rows[2], left, src, w, edges);
927
6.69k
        left++;
928
6.69k
        src += PXSTRIDE(stride);
929
930
6.69k
        if (--h <= 0)
931
799
            goto odd;
932
933
5.89k
        sgr_box5_row_h(sumsq_rows[3], sum_rows[3], left, src, w, edges);
934
5.89k
        left++;
935
5.89k
        src += PXSTRIDE(stride);
936
937
5.89k
        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
938
5.89k
                      w, params->sgr.s0, BITDEPTH_MAX);
939
5.89k
        sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
940
5.89k
                    w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
941
942
5.89k
        if (--h <= 0)
943
439
            goto vert_2;
944
945
        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
946
        // one of them to point at the previously unused rows[4].
947
5.45k
        sumsq_ptrs[3] = sumsq_rows[4];
948
5.45k
        sum_ptrs[3] = sum_rows[4];
949
5.45k
    }
950
951
497k
    do {
952
497k
        sgr_box5_row_h(sumsq_ptrs[3], sum_ptrs[3], left, src, w, edges);
953
497k
        left++;
954
497k
        src += PXSTRIDE(stride);
955
956
497k
        if (--h <= 0)
957
3.02k
            goto odd;
958
959
494k
        sgr_box5_row_h(sumsq_ptrs[4], sum_ptrs[4], left, src, w, edges);
960
494k
        left++;
961
494k
        src += PXSTRIDE(stride);
962
963
494k
        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
964
494k
                      w, params->sgr.s0, BITDEPTH_MAX);
965
494k
        sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
966
494k
                    w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
967
494k
    } while (--h > 0);
968
969
18.8k
    if (!(edges & LR_HAVE_BOTTOM))
970
2.80k
        goto vert_2;
971
972
16.0k
    sgr_box5_row_h(sumsq_ptrs[3], sum_ptrs[3], NULL, lpf_bottom, w, edges);
973
16.0k
    lpf_bottom += PXSTRIDE(stride);
974
16.0k
    sgr_box5_row_h(sumsq_ptrs[4], sum_ptrs[4], NULL, lpf_bottom, w, edges);
975
976
20.7k
output_2:
977
20.7k
    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
978
20.7k
                  w, params->sgr.s0, BITDEPTH_MAX);
979
20.7k
    sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
980
20.7k
                w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
981
20.7k
    return;
982
983
4.72k
vert_2:
984
    // Duplicate the last row twice more
985
4.72k
    sumsq_ptrs[3] = sumsq_ptrs[2];
986
4.72k
    sumsq_ptrs[4] = sumsq_ptrs[2];
987
4.72k
    sum_ptrs[3] = sum_ptrs[2];
988
4.72k
    sum_ptrs[4] = sum_ptrs[2];
989
4.72k
    goto output_2;
990
991
3.82k
odd:
992
    // Copy the last row as padding once
993
3.82k
    sumsq_ptrs[4] = sumsq_ptrs[3];
994
3.82k
    sum_ptrs[4] = sum_ptrs[3];
995
996
3.82k
    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
997
3.82k
                  w, params->sgr.s0, BITDEPTH_MAX);
998
3.82k
    sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
999
3.82k
                w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
1000
1001
5.34k
output_1:
1002
    // Duplicate the last row twice more
1003
5.34k
    sumsq_ptrs[3] = sumsq_ptrs[2];
1004
5.34k
    sumsq_ptrs[4] = sumsq_ptrs[2];
1005
5.34k
    sum_ptrs[3] = sum_ptrs[2];
1006
5.34k
    sum_ptrs[4] = sum_ptrs[2];
1007
1008
5.34k
    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
1009
5.34k
                  w, params->sgr.s0, BITDEPTH_MAX);
1010
    // Output only one row
1011
5.34k
    sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
1012
5.34k
                w, 1, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
1013
5.34k
    return;
1014
1015
1.52k
vert_1:
1016
    // Copy the last row as padding once
1017
1.52k
    sumsq_ptrs[4] = sumsq_ptrs[3];
1018
1.52k
    sum_ptrs[4] = sum_ptrs[3];
1019
1020
1.52k
    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
1021
1.52k
                  w, params->sgr.s0, BITDEPTH_MAX);
1022
1.52k
    rotate(A_ptrs, B_ptrs, 2);
1023
1024
1.52k
    goto output_1;
1025
3.82k
}
1026
1027
static void sgr_mix_c(pixel *dst, const ptrdiff_t stride,
1028
                      const pixel (*left)[4], const pixel *lpf,
1029
                      const int w, int h,
1030
                      const LooprestorationParams *const params,
1031
                      const enum LrEdgeFlags edges HIGHBD_DECL_SUFFIX)
1032
56.0k
{
1033
56.0k
    ALIGN_STK_16(int32_t, sumsq5_buf, BUF_STRIDE * 5 + 16,);
1034
56.0k
    ALIGN_STK_16(coef, sum5_buf, BUF_STRIDE * 5 + 16,);
1035
56.0k
    int32_t *sumsq5_ptrs[5], *sumsq5_rows[5];
1036
56.0k
    coef *sum5_ptrs[5], *sum5_rows[5];
1037
336k
    for (int i = 0; i < 5; i++) {
1038
280k
        sumsq5_rows[i] = &sumsq5_buf[i * BUF_STRIDE];
1039
280k
        sum5_rows[i] = &sum5_buf[i * BUF_STRIDE];
1040
280k
    }
1041
56.0k
    ALIGN_STK_16(int32_t, sumsq3_buf, BUF_STRIDE * 3 + 16,);
1042
56.0k
    ALIGN_STK_16(coef, sum3_buf, BUF_STRIDE * 3 + 16,);
1043
56.0k
    int32_t *sumsq3_ptrs[3], *sumsq3_rows[3];
1044
56.0k
    coef *sum3_ptrs[3], *sum3_rows[3];
1045
224k
    for (int i = 0; i < 3; i++) {
1046
168k
        sumsq3_rows[i] = &sumsq3_buf[i * BUF_STRIDE];
1047
168k
        sum3_rows[i] = &sum3_buf[i * BUF_STRIDE];
1048
168k
    }
1049
1050
56.0k
    ALIGN_STK_16(int32_t, A5_buf, BUF_STRIDE * 2 + 16,);
1051
56.0k
    ALIGN_STK_16(coef, B5_buf, BUF_STRIDE * 2 + 16,);
1052
56.0k
    int32_t *A5_ptrs[2];
1053
56.0k
    coef *B5_ptrs[2];
1054
168k
    for (int i = 0; i < 2; i++) {
1055
112k
        A5_ptrs[i] = &A5_buf[i * BUF_STRIDE];
1056
112k
        B5_ptrs[i] = &B5_buf[i * BUF_STRIDE];
1057
112k
    }
1058
56.0k
    ALIGN_STK_16(int32_t, A3_buf, BUF_STRIDE * 4 + 16,);
1059
56.0k
    ALIGN_STK_16(coef, B3_buf, BUF_STRIDE * 4 + 16,);
1060
56.0k
    int32_t *A3_ptrs[4];
1061
56.0k
    coef *B3_ptrs[4];
1062
280k
    for (int i = 0; i < 4; i++) {
1063
224k
        A3_ptrs[i] = &A3_buf[i * BUF_STRIDE];
1064
224k
        B3_ptrs[i] = &B3_buf[i * BUF_STRIDE];
1065
224k
    }
1066
56.0k
    const pixel *src = dst;
1067
56.0k
    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
1068
1069
56.0k
    if (edges & LR_HAVE_TOP) {
1070
37.2k
        sumsq5_ptrs[0] = sumsq5_rows[0];
1071
37.2k
        sumsq5_ptrs[1] = sumsq5_rows[0];
1072
37.2k
        sumsq5_ptrs[2] = sumsq5_rows[1];
1073
37.2k
        sumsq5_ptrs[3] = sumsq5_rows[2];
1074
37.2k
        sumsq5_ptrs[4] = sumsq5_rows[3];
1075
37.2k
        sum5_ptrs[0] = sum5_rows[0];
1076
37.2k
        sum5_ptrs[1] = sum5_rows[0];
1077
37.2k
        sum5_ptrs[2] = sum5_rows[1];
1078
37.2k
        sum5_ptrs[3] = sum5_rows[2];
1079
37.2k
        sum5_ptrs[4] = sum5_rows[3];
1080
1081
37.2k
        sumsq3_ptrs[0] = sumsq3_rows[0];
1082
37.2k
        sumsq3_ptrs[1] = sumsq3_rows[1];
1083
37.2k
        sumsq3_ptrs[2] = sumsq3_rows[2];
1084
37.2k
        sum3_ptrs[0] = sum3_rows[0];
1085
37.2k
        sum3_ptrs[1] = sum3_rows[1];
1086
37.2k
        sum3_ptrs[2] = sum3_rows[2];
1087
1088
37.2k
        sgr_box35_row_h(sumsq3_rows[0], sum3_rows[0],
1089
37.2k
                        sumsq5_rows[0], sum5_rows[0],
1090
37.2k
                        NULL, lpf, w, edges);
1091
37.2k
        lpf += PXSTRIDE(stride);
1092
37.2k
        sgr_box35_row_h(sumsq3_rows[1], sum3_rows[1],
1093
37.2k
                        sumsq5_rows[1], sum5_rows[1],
1094
37.2k
                        NULL, lpf, w, edges);
1095
1096
37.2k
        sgr_box35_row_h(sumsq3_rows[2], sum3_rows[2],
1097
37.2k
                        sumsq5_rows[2], sum5_rows[2],
1098
37.2k
                        left, src, w, edges);
1099
37.2k
        left++;
1100
37.2k
        src += PXSTRIDE(stride);
1101
1102
37.2k
        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1103
37.2k
                      w, params->sgr.s1, BITDEPTH_MAX);
1104
37.2k
        rotate(A3_ptrs, B3_ptrs, 4);
1105
1106
37.2k
        if (--h <= 0)
1107
777
            goto vert_1;
1108
1109
36.4k
        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
1110
36.4k
                        sumsq5_rows[3], sum5_rows[3],
1111
36.4k
                        left, src, w, edges);
1112
36.4k
        left++;
1113
36.4k
        src += PXSTRIDE(stride);
1114
36.4k
        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
1115
36.4k
                      w, params->sgr.s0, BITDEPTH_MAX);
1116
36.4k
        rotate(A5_ptrs, B5_ptrs, 2);
1117
36.4k
        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1118
36.4k
                      w, params->sgr.s1, BITDEPTH_MAX);
1119
36.4k
        rotate(A3_ptrs, B3_ptrs, 4);
1120
1121
36.4k
        if (--h <= 0)
1122
499
            goto vert_2;
1123
1124
        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
1125
        // one of them to point at the previously unused rows[4].
1126
35.9k
        sumsq5_ptrs[3] = sumsq5_rows[4];
1127
35.9k
        sum5_ptrs[3] = sum5_rows[4];
1128
35.9k
    } else {
1129
18.7k
        sumsq5_ptrs[0] = sumsq5_rows[0];
1130
18.7k
        sumsq5_ptrs[1] = sumsq5_rows[0];
1131
18.7k
        sumsq5_ptrs[2] = sumsq5_rows[0];
1132
18.7k
        sumsq5_ptrs[3] = sumsq5_rows[0];
1133
18.7k
        sumsq5_ptrs[4] = sumsq5_rows[0];
1134
18.7k
        sum5_ptrs[0] = sum5_rows[0];
1135
18.7k
        sum5_ptrs[1] = sum5_rows[0];
1136
18.7k
        sum5_ptrs[2] = sum5_rows[0];
1137
18.7k
        sum5_ptrs[3] = sum5_rows[0];
1138
18.7k
        sum5_ptrs[4] = sum5_rows[0];
1139
1140
18.7k
        sumsq3_ptrs[0] = sumsq3_rows[0];
1141
18.7k
        sumsq3_ptrs[1] = sumsq3_rows[0];
1142
18.7k
        sumsq3_ptrs[2] = sumsq3_rows[0];
1143
18.7k
        sum3_ptrs[0] = sum3_rows[0];
1144
18.7k
        sum3_ptrs[1] = sum3_rows[0];
1145
18.7k
        sum3_ptrs[2] = sum3_rows[0];
1146
1147
18.7k
        sgr_box35_row_h(sumsq3_rows[0], sum3_rows[0],
1148
18.7k
                        sumsq5_rows[0], sum5_rows[0],
1149
18.7k
                        left, src, w, edges);
1150
18.7k
        left++;
1151
18.7k
        src += PXSTRIDE(stride);
1152
1153
18.7k
        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1154
18.7k
                      w, params->sgr.s1, BITDEPTH_MAX);
1155
18.7k
        rotate(A3_ptrs, B3_ptrs, 4);
1156
1157
18.7k
        if (--h <= 0)
1158
3.29k
            goto vert_1;
1159
1160
15.4k
        sumsq5_ptrs[4] = sumsq5_rows[1];
1161
15.4k
        sum5_ptrs[4] = sum5_rows[1];
1162
1163
15.4k
        sumsq3_ptrs[2] = sumsq3_rows[1];
1164
15.4k
        sum3_ptrs[2] = sum3_rows[1];
1165
1166
15.4k
        sgr_box35_row_h(sumsq3_rows[1], sum3_rows[1],
1167
15.4k
                        sumsq5_rows[1], sum5_rows[1],
1168
15.4k
                        left, src, w, edges);
1169
15.4k
        left++;
1170
15.4k
        src += PXSTRIDE(stride);
1171
1172
15.4k
        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
1173
15.4k
                      w, params->sgr.s0, BITDEPTH_MAX);
1174
15.4k
        rotate(A5_ptrs, B5_ptrs, 2);
1175
15.4k
        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1176
15.4k
                      w, params->sgr.s1, BITDEPTH_MAX);
1177
15.4k
        rotate(A3_ptrs, B3_ptrs, 4);
1178
1179
15.4k
        if (--h <= 0)
1180
2.91k
            goto vert_2;
1181
1182
12.5k
        sumsq5_ptrs[3] = sumsq5_rows[2];
1183
12.5k
        sumsq5_ptrs[4] = sumsq5_rows[3];
1184
12.5k
        sum5_ptrs[3] = sum5_rows[2];
1185
12.5k
        sum5_ptrs[4] = sum5_rows[3];
1186
1187
12.5k
        sumsq3_ptrs[2] = sumsq3_rows[2];
1188
12.5k
        sum3_ptrs[2] = sum3_rows[2];
1189
1190
12.5k
        sgr_box35_row_h(sumsq3_rows[2], sum3_rows[2],
1191
12.5k
                        sumsq5_rows[2], sum5_rows[2],
1192
12.5k
                        left, src, w, edges);
1193
12.5k
        left++;
1194
12.5k
        src += PXSTRIDE(stride);
1195
1196
12.5k
        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1197
12.5k
                      w, params->sgr.s1, BITDEPTH_MAX);
1198
12.5k
        rotate(A3_ptrs, B3_ptrs, 4);
1199
1200
12.5k
        if (--h <= 0)
1201
1.44k
            goto odd;
1202
1203
11.1k
        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
1204
11.1k
                        sumsq5_rows[3], sum5_rows[3],
1205
11.1k
                        left, src, w, edges);
1206
11.1k
        left++;
1207
11.1k
        src += PXSTRIDE(stride);
1208
1209
11.1k
        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
1210
11.1k
                      w, params->sgr.s0, BITDEPTH_MAX);
1211
11.1k
        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1212
11.1k
                      w, params->sgr.s1, BITDEPTH_MAX);
1213
11.1k
        sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
1214
11.1k
                       w, 2, params->sgr.w0, params->sgr.w1
1215
11.1k
                       HIGHBD_TAIL_SUFFIX);
1216
1217
11.1k
        if (--h <= 0)
1218
683
            goto vert_2;
1219
1220
        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
1221
        // one of them to point at the previously unused rows[4].
1222
10.4k
        sumsq5_ptrs[3] = sumsq5_rows[4];
1223
10.4k
        sum5_ptrs[3] = sum5_rows[4];
1224
10.4k
    }
1225
1226
1.01M
    do {
1227
1.01M
        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
1228
1.01M
                        sumsq5_ptrs[3], sum5_ptrs[3],
1229
1.01M
                        left, src, w, edges);
1230
1.01M
        left++;
1231
1.01M
        src += PXSTRIDE(stride);
1232
1233
1.01M
        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1234
1.01M
                      w, params->sgr.s1, BITDEPTH_MAX);
1235
1.01M
        rotate(A3_ptrs, B3_ptrs, 4);
1236
1237
1.01M
        if (--h <= 0)
1238
8.07k
            goto odd;
1239
1240
1.00M
        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
1241
1.00M
                        sumsq5_ptrs[4], sum5_ptrs[4],
1242
1.00M
                        left, src, w, edges);
1243
1.00M
        left++;
1244
1.00M
        src += PXSTRIDE(stride);
1245
1246
1.00M
        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
1247
1.00M
                      w, params->sgr.s0, BITDEPTH_MAX);
1248
1.00M
        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1249
1.00M
                      w, params->sgr.s1, BITDEPTH_MAX);
1250
1.00M
        sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
1251
1.00M
                       w, 2, params->sgr.w0, params->sgr.w1
1252
1.00M
                       HIGHBD_TAIL_SUFFIX);
1253
1.00M
    } while (--h > 0);
1254
1255
38.3k
    if (!(edges & LR_HAVE_BOTTOM))
1256
3.85k
        goto vert_2;
1257
1258
34.4k
    sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
1259
34.4k
                    sumsq5_ptrs[3], sum5_ptrs[3],
1260
34.4k
                    NULL, lpf_bottom, w, edges);
1261
34.4k
    lpf_bottom += PXSTRIDE(stride);
1262
34.4k
    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1263
34.4k
                  w, params->sgr.s1, BITDEPTH_MAX);
1264
34.4k
    rotate(A3_ptrs, B3_ptrs, 4);
1265
1266
34.4k
    sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
1267
34.4k
                    sumsq5_ptrs[4], sum5_ptrs[4],
1268
34.4k
                    NULL, lpf_bottom, w, edges);
1269
1270
42.4k
output_2:
1271
42.4k
    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
1272
42.4k
                  w, params->sgr.s0, BITDEPTH_MAX);
1273
42.4k
    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1274
42.4k
                  w, params->sgr.s1, BITDEPTH_MAX);
1275
42.4k
    sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
1276
42.4k
                   w, 2, params->sgr.w0, params->sgr.w1
1277
42.4k
                   HIGHBD_TAIL_SUFFIX);
1278
42.4k
    return;
1279
1280
7.94k
vert_2:
1281
    // Duplicate the last row twice more
1282
7.94k
    sumsq5_ptrs[3] = sumsq5_ptrs[2];
1283
7.94k
    sumsq5_ptrs[4] = sumsq5_ptrs[2];
1284
7.94k
    sum5_ptrs[3] = sum5_ptrs[2];
1285
7.94k
    sum5_ptrs[4] = sum5_ptrs[2];
1286
1287
7.94k
    sumsq3_ptrs[2] = sumsq3_ptrs[1];
1288
7.94k
    sum3_ptrs[2] = sum3_ptrs[1];
1289
7.94k
    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1290
7.94k
                  w, params->sgr.s1, BITDEPTH_MAX);
1291
7.94k
    rotate(A3_ptrs, B3_ptrs, 4);
1292
1293
7.94k
    sumsq3_ptrs[2] = sumsq3_ptrs[1];
1294
7.94k
    sum3_ptrs[2] = sum3_ptrs[1];
1295
1296
7.94k
    goto output_2;
1297
1298
9.51k
odd:
1299
    // Copy the last row as padding once
1300
9.51k
    sumsq5_ptrs[4] = sumsq5_ptrs[3];
1301
9.51k
    sum5_ptrs[4] = sum5_ptrs[3];
1302
1303
9.51k
    sumsq3_ptrs[2] = sumsq3_ptrs[1];
1304
9.51k
    sum3_ptrs[2] = sum3_ptrs[1];
1305
1306
9.51k
    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
1307
9.51k
                  w, params->sgr.s0, BITDEPTH_MAX);
1308
9.51k
    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1309
9.51k
                  w, params->sgr.s1, BITDEPTH_MAX);
1310
9.51k
    sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
1311
9.51k
                   w, 2, params->sgr.w0, params->sgr.w1
1312
9.51k
                   HIGHBD_TAIL_SUFFIX);
1313
1314
13.5k
output_1:
1315
    // Duplicate the last row twice more
1316
13.5k
    sumsq5_ptrs[3] = sumsq5_ptrs[2];
1317
13.5k
    sumsq5_ptrs[4] = sumsq5_ptrs[2];
1318
13.5k
    sum5_ptrs[3] = sum5_ptrs[2];
1319
13.5k
    sum5_ptrs[4] = sum5_ptrs[2];
1320
1321
13.5k
    sumsq3_ptrs[2] = sumsq3_ptrs[1];
1322
13.5k
    sum3_ptrs[2] = sum3_ptrs[1];
1323
1324
13.5k
    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
1325
13.5k
                  w, params->sgr.s0, BITDEPTH_MAX);
1326
13.5k
    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1327
13.5k
                  w, params->sgr.s1, BITDEPTH_MAX);
1328
13.5k
    rotate(A3_ptrs, B3_ptrs, 4);
1329
    // Output only one row
1330
13.5k
    sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
1331
13.5k
                   w, 1, params->sgr.w0, params->sgr.w1
1332
13.5k
                   HIGHBD_TAIL_SUFFIX);
1333
13.5k
    return;
1334
1335
4.06k
vert_1:
1336
    // Copy the last row as padding once
1337
4.06k
    sumsq5_ptrs[4] = sumsq5_ptrs[3];
1338
4.06k
    sum5_ptrs[4] = sum5_ptrs[3];
1339
1340
4.06k
    sumsq3_ptrs[2] = sumsq3_ptrs[1];
1341
4.06k
    sum3_ptrs[2] = sum3_ptrs[1];
1342
1343
4.06k
    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
1344
4.06k
                  w, params->sgr.s0, BITDEPTH_MAX);
1345
4.06k
    rotate(A5_ptrs, B5_ptrs, 2);
1346
4.06k
    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
1347
4.06k
                  w, params->sgr.s1, BITDEPTH_MAX);
1348
4.06k
    rotate(A3_ptrs, B3_ptrs, 4);
1349
1350
4.06k
    goto output_1;
1351
9.51k
}
1352
1353
#if HAVE_ASM
1354
#if ARCH_AARCH64 || ARCH_ARM
1355
#include "src/arm/looprestoration.h"
1356
#elif ARCH_LOONGARCH64
1357
#include "src/loongarch/looprestoration.h"
1358
#elif ARCH_PPC64LE
1359
#include "src/ppc/looprestoration.h"
1360
#elif ARCH_X86
1361
#include "src/x86/looprestoration.h"
1362
#endif
1363
#endif
1364
1365
COLD void bitfn(dav1d_loop_restoration_dsp_init)(Dav1dLoopRestorationDSPContext *const c,
1366
                                                 const int bpc)
1367
55.8k
{
1368
55.8k
    c->wiener[0] = c->wiener[1] = wiener_c;
1369
55.8k
    c->sgr[0] = sgr_5x5_c;
1370
55.8k
    c->sgr[1] = sgr_3x3_c;
1371
55.8k
    c->sgr[2] = sgr_mix_c;
1372
1373
#if HAVE_ASM
1374
#if ARCH_AARCH64 || ARCH_ARM
1375
    loop_restoration_dsp_init_arm(c, bpc);
1376
#elif ARCH_LOONGARCH64
1377
    loop_restoration_dsp_init_loongarch(c, bpc);
1378
#elif ARCH_PPC64LE
1379
    loop_restoration_dsp_init_ppc(c, bpc);
1380
#elif ARCH_X86
1381
    loop_restoration_dsp_init_x86(c, bpc);
1382
#endif
1383
#endif
1384
55.8k
}
dav1d_loop_restoration_dsp_init_8bpc
Line
Count
Source
1367
25.1k
{
1368
25.1k
    c->wiener[0] = c->wiener[1] = wiener_c;
1369
25.1k
    c->sgr[0] = sgr_5x5_c;
1370
25.1k
    c->sgr[1] = sgr_3x3_c;
1371
25.1k
    c->sgr[2] = sgr_mix_c;
1372
1373
#if HAVE_ASM
1374
#if ARCH_AARCH64 || ARCH_ARM
1375
    loop_restoration_dsp_init_arm(c, bpc);
1376
#elif ARCH_LOONGARCH64
1377
    loop_restoration_dsp_init_loongarch(c, bpc);
1378
#elif ARCH_PPC64LE
1379
    loop_restoration_dsp_init_ppc(c, bpc);
1380
#elif ARCH_X86
1381
    loop_restoration_dsp_init_x86(c, bpc);
1382
#endif
1383
#endif
1384
25.1k
}
dav1d_loop_restoration_dsp_init_16bpc
Line
Count
Source
1367
30.6k
{
1368
30.6k
    c->wiener[0] = c->wiener[1] = wiener_c;
1369
30.6k
    c->sgr[0] = sgr_5x5_c;
1370
30.6k
    c->sgr[1] = sgr_3x3_c;
1371
30.6k
    c->sgr[2] = sgr_mix_c;
1372
1373
#if HAVE_ASM
1374
#if ARCH_AARCH64 || ARCH_ARM
1375
    loop_restoration_dsp_init_arm(c, bpc);
1376
#elif ARCH_LOONGARCH64
1377
    loop_restoration_dsp_init_loongarch(c, bpc);
1378
#elif ARCH_PPC64LE
1379
    loop_restoration_dsp_init_ppc(c, bpc);
1380
#elif ARCH_X86
1381
    loop_restoration_dsp_init_x86(c, bpc);
1382
#endif
1383
#endif
1384
30.6k
}