Coverage Report

Created: 2026-07-30 06:27

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/dav1d/src/refmvs.c
Line
Count
Source
1
/*
2
 * Copyright © 2020, VideoLAN and dav1d authors
3
 * Copyright © 2020, Two Orioles, LLC
4
 * All rights reserved.
5
 *
6
 * Redistribution and use in source and binary forms, with or without
7
 * modification, are permitted provided that the following conditions are met:
8
 *
9
 * 1. Redistributions of source code must retain the above copyright notice, this
10
 *    list of conditions and the following disclaimer.
11
 *
12
 * 2. Redistributions in binary form must reproduce the above copyright notice,
13
 *    this list of conditions and the following disclaimer in the documentation
14
 *    and/or other materials provided with the distribution.
15
 *
16
 * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
17
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
18
 * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
19
 * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
20
 * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
21
 * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
22
 * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
23
 * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
24
 * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
25
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
26
 */
27
28
#include "config.h"
29
30
#include <limits.h>
31
#include <stdlib.h>
32
33
#include "dav1d/common.h"
34
35
#include "common/intops.h"
36
37
#include "src/env.h"
38
#include "src/mem.h"
39
#include "src/refmvs.h"
40
41
static void add_spatial_candidate(refmvs_candidate *const mvstack, int *const cnt,
42
                                  const int weight, const refmvs_block *const b,
43
                                  const union refmvs_refpair ref, const mv gmv[2],
44
                                  int *const have_newmv_match,
45
                                  int *const have_refmv_match)
46
7.65M
{
47
7.65M
    if (b->mv.mv[0].n == INVALID_MV) return; // intra block, no intrabc
48
49
6.61M
    if (ref.ref[1] == -1) {
50
7.55M
        for (int n = 0; n < 2; n++) {
51
6.81M
            if (b->ref.ref[n] == ref.ref[0]) {
52
5.24M
                const mv cand_mv = ((b->mf & 1) && gmv[0].n != INVALID_MV) ?
53
4.96M
                                   gmv[0] : b->mv.mv[n];
54
55
5.24M
                *have_refmv_match = 1;
56
5.24M
                *have_newmv_match |= b->mf >> 1;
57
58
5.24M
                const int last = *cnt;
59
8.90M
                for (int m = 0; m < last; m++)
60
5.88M
                    if (mvstack[m].mv.mv[0].n == cand_mv.n) {
61
2.21M
                        mvstack[m].weight += weight;
62
2.21M
                        return;
63
2.21M
                    }
64
65
3.02M
                if (last < 8) {
66
3.01M
                    mvstack[last].mv.mv[0] = cand_mv;
67
3.01M
                    mvstack[last].weight = weight;
68
3.01M
                    *cnt = last + 1;
69
3.01M
                }
70
3.02M
                return;
71
5.24M
            }
72
6.81M
        }
73
5.98M
    } else if (b->ref.pair == ref.pair) {
74
227k
        const refmvs_mvpair cand_mv = { .mv = {
75
227k
            [0] = ((b->mf & 1) && gmv[0].n != INVALID_MV) ? gmv[0] : b->mv.mv[0],
76
227k
            [1] = ((b->mf & 1) && gmv[1].n != INVALID_MV) ? gmv[1] : b->mv.mv[1],
77
227k
        }};
78
79
227k
        *have_refmv_match = 1;
80
227k
        *have_newmv_match |= b->mf >> 1;
81
82
227k
        const int last = *cnt;
83
366k
        for (int n = 0; n < last; n++)
84
213k
            if (mvstack[n].mv.n == cand_mv.n) {
85
74.5k
                mvstack[n].weight += weight;
86
74.5k
                return;
87
74.5k
            }
88
89
153k
        if (last < 8) {
90
153k
            mvstack[last].mv = cand_mv;
91
153k
            mvstack[last].weight = weight;
92
153k
            *cnt = last + 1;
93
153k
        }
94
153k
    }
95
6.61M
}
96
97
static int scan_row(refmvs_candidate *const mvstack, int *const cnt,
98
                    const union refmvs_refpair ref, const mv gmv[2],
99
                    const refmvs_block *b, const int bw4, const int w4,
100
                    const int max_rows, const int step,
101
                    int *const have_newmv_match, int *const have_refmv_match)
102
2.32M
{
103
2.32M
    const refmvs_block *cand_b = b;
104
2.32M
    const enum BlockSize first_cand_bs = cand_b->bs;
105
2.32M
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
106
2.32M
    int cand_bw4 = first_cand_b_dim[0];
107
2.32M
    int len = imax(step, imin(bw4, cand_bw4));
108
109
2.32M
    if (bw4 <= cand_bw4) {
110
        // FIXME weight can be higher for odd blocks (bx4 & 1), but then the
111
        // position of the first block has to be odd already, i.e. not just
112
        // for row_offset=-3/-5
113
        // FIXME why can this not be cand_bw4?
114
2.03M
        const int weight = bw4 == 1 ? 2 :
115
2.03M
                           imax(2, imin(2 * max_rows, first_cand_b_dim[1]));
116
2.03M
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
117
2.03M
                              have_newmv_match, have_refmv_match);
118
2.03M
        return weight >> 1;
119
2.03M
    }
120
121
578k
    for (int x = 0;;) {
122
        // FIXME if we overhang above, we could fill a bitmask so we don't have
123
        // to repeat the add_spatial_candidate() for the next row, but just increase
124
        // the weight here
125
578k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
126
578k
                              have_newmv_match, have_refmv_match);
127
578k
        x += len;
128
578k
        if (x >= w4) return 1;
129
284k
        cand_b = &b[x];
130
284k
        cand_bw4 = dav1d_block_dimensions[cand_b->bs][0];
131
284k
        assert(cand_bw4 < bw4);
132
284k
        len = imax(step, cand_bw4);
133
284k
    }
134
294k
}
135
136
static int scan_col(refmvs_candidate *const mvstack, int *const cnt,
137
                    const union refmvs_refpair ref, const mv gmv[2],
138
                    /*const*/ refmvs_block *const *b, const int bh4, const int h4,
139
                    const int bx4, const int max_cols, const int step,
140
                    int *const have_newmv_match, int *const have_refmv_match)
141
3.05M
{
142
3.05M
    const refmvs_block *cand_b = &b[0][bx4];
143
3.05M
    const enum BlockSize first_cand_bs = cand_b->bs;
144
3.05M
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
145
3.05M
    int cand_bh4 = first_cand_b_dim[1];
146
3.05M
    int len = imax(step, imin(bh4, cand_bh4));
147
148
3.05M
    if (bh4 <= cand_bh4) {
149
        // FIXME weight can be higher for odd blocks (by4 & 1), but then the
150
        // position of the first block has to be odd already, i.e. not just
151
        // for col_offset=-3/-5
152
        // FIXME why can this not be cand_bh4?
153
2.69M
        const int weight = bh4 == 1 ? 2 :
154
2.69M
                           imax(2, imin(2 * max_cols, first_cand_b_dim[0]));
155
2.69M
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
156
2.69M
                            have_newmv_match, have_refmv_match);
157
2.69M
        return weight >> 1;
158
2.69M
    }
159
160
690k
    for (int y = 0;;) {
161
        // FIXME if we overhang above, we could fill a bitmask so we don't have
162
        // to repeat the add_spatial_candidate() for the next row, but just increase
163
        // the weight here
164
690k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
165
690k
                              have_newmv_match, have_refmv_match);
166
690k
        y += len;
167
690k
        if (y >= h4) return 1;
168
337k
        cand_b = &b[y][bx4];
169
337k
        cand_bh4 = dav1d_block_dimensions[cand_b->bs][1];
170
337k
        assert(cand_bh4 < bh4);
171
337k
        len = imax(step, cand_bh4);
172
337k
    }
173
353k
}
174
175
136k
static inline union mv mv_projection(const union mv mv, const int num, const int den) {
176
136k
    static const uint16_t div_mult[32] = {
177
136k
           0, 16384, 8192, 5461, 4096, 3276, 2730, 2340,
178
136k
        2048,  1820, 1638, 1489, 1365, 1260, 1170, 1092,
179
136k
        1024,   963,  910,  862,  819,  780,  744,  712,
180
136k
         682,   655,  630,  606,  585,  564,  546,  528
181
136k
    };
182
136k
    assert(den > 0 && den < 32);
183
136k
    assert(num > -32 && num < 32);
184
136k
    const int frac = num * div_mult[den];
185
136k
    const int y = mv.y * frac, x = mv.x * frac;
186
    // Round and clip according to AV1 spec section 7.9.3
187
136k
    return (union mv) { // 0x3fff == (1 << 14) - 1
188
136k
        .y = iclip((y + 8192 + (y >> 31)) >> 14, -0x3fff, 0x3fff),
189
136k
        .x = iclip((x + 8192 + (x >> 31)) >> 14, -0x3fff, 0x3fff)
190
136k
    };
191
136k
}
192
193
static void add_temporal_candidate(const refmvs_frame *const rf,
194
                                   refmvs_candidate *const mvstack, int *const cnt,
195
                                   const refmvs_temporal_block *const rb,
196
                                   const union refmvs_refpair ref, int *const globalmv_ctx,
197
                                   const union mv gmv[])
198
136k
{
199
136k
    if (rb->mv.n == INVALID_MV) return;
200
201
72.7k
    union mv mv = mv_projection(rb->mv, rf->pocdiff[ref.ref[0] - 1], rb->ref);
202
72.7k
    fix_mv_precision(rf->frm_hdr, &mv);
203
204
72.7k
    const int last = *cnt;
205
72.7k
    if (ref.ref[1] == -1) {
206
51.0k
        if (globalmv_ctx)
207
11.4k
            *globalmv_ctx = (abs(mv.x - gmv[0].x) | abs(mv.y - gmv[0].y)) >= 16;
208
209
78.2k
        for (int n = 0; n < last; n++)
210
68.1k
            if (mvstack[n].mv.mv[0].n == mv.n) {
211
40.9k
                mvstack[n].weight += 2;
212
40.9k
                return;
213
40.9k
            }
214
10.0k
        if (last < 8) {
215
9.99k
            mvstack[last].mv.mv[0] = mv;
216
9.99k
            mvstack[last].weight = 2;
217
9.99k
            *cnt = last + 1;
218
9.99k
        }
219
21.7k
    } else {
220
21.7k
        refmvs_mvpair mvp = { .mv = {
221
21.7k
            [0] = mv,
222
21.7k
            [1] = mv_projection(rb->mv, rf->pocdiff[ref.ref[1] - 1], rb->ref),
223
21.7k
        }};
224
21.7k
        fix_mv_precision(rf->frm_hdr, &mvp.mv[1]);
225
226
32.9k
        for (int n = 0; n < last; n++)
227
27.7k
            if (mvstack[n].mv.n == mvp.n) {
228
16.5k
                mvstack[n].weight += 2;
229
16.5k
                return;
230
16.5k
            }
231
5.14k
        if (last < 8) {
232
5.11k
            mvstack[last].mv = mvp;
233
5.11k
            mvstack[last].weight = 2;
234
5.11k
            *cnt = last + 1;
235
5.11k
        }
236
5.14k
    }
237
72.7k
}
238
239
static void add_compound_extended_candidate(refmvs_candidate *const same,
240
                                            int *const same_count,
241
                                            const refmvs_block *const cand_b,
242
                                            const int sign0, const int sign1,
243
                                            const union refmvs_refpair ref,
244
                                            const uint8_t *const sign_bias)
245
164k
{
246
164k
    refmvs_candidate *const diff = &same[2];
247
164k
    int *const diff_count = &same_count[2];
248
249
420k
    for (int n = 0; n < 2; n++) {
250
323k
        const int cand_ref = cand_b->ref.ref[n];
251
252
323k
        if (cand_ref <= 0) break;
253
254
255k
        mv cand_mv = cand_b->mv.mv[n];
255
255k
        if (cand_ref == ref.ref[0]) {
256
89.8k
            if (same_count[0] < 2)
257
87.2k
                same[same_count[0]++].mv.mv[0] = cand_mv;
258
89.8k
            if (diff_count[1] < 2) {
259
78.1k
                if (sign1 ^ sign_bias[cand_ref - 1]) {
260
2.80k
                    cand_mv.y = -cand_mv.y;
261
2.80k
                    cand_mv.x = -cand_mv.x;
262
2.80k
                }
263
78.1k
                diff[diff_count[1]++].mv.mv[1] = cand_mv;
264
78.1k
            }
265
166k
        } else if (cand_ref == ref.ref[1]) {
266
90.5k
            if (same_count[1] < 2)
267
88.6k
                same[same_count[1]++].mv.mv[1] = cand_mv;
268
90.5k
            if (diff_count[0] < 2) {
269
75.4k
                if (sign0 ^ sign_bias[cand_ref - 1]) {
270
2.87k
                    cand_mv.y = -cand_mv.y;
271
2.87k
                    cand_mv.x = -cand_mv.x;
272
2.87k
                }
273
75.4k
                diff[diff_count[0]++].mv.mv[0] = cand_mv;
274
75.4k
            }
275
90.5k
        } else {
276
75.5k
            mv i_cand_mv = (union mv) {
277
75.5k
                .x = -cand_mv.x,
278
75.5k
                .y = -cand_mv.y
279
75.5k
            };
280
281
75.5k
            if (diff_count[0] < 2) {
282
60.0k
                diff[diff_count[0]++].mv.mv[0] =
283
60.0k
                    sign0 ^ sign_bias[cand_ref - 1] ?
284
58.8k
                    i_cand_mv : cand_mv;
285
60.0k
            }
286
287
75.5k
            if (diff_count[1] < 2) {
288
57.2k
                diff[diff_count[1]++].mv.mv[1] =
289
57.2k
                    sign1 ^ sign_bias[cand_ref - 1] ?
290
56.1k
                    i_cand_mv : cand_mv;
291
57.2k
            }
292
75.5k
        }
293
255k
    }
294
164k
}
295
296
static void add_single_extended_candidate(refmvs_candidate mvstack[8], int *const cnt,
297
                                          const refmvs_block *const cand_b,
298
                                          const int sign, const uint8_t *const sign_bias)
299
908k
{
300
1.82M
    for (int n = 0; n < 2; n++) {
301
1.78M
        const int cand_ref = cand_b->ref.ref[n];
302
303
1.78M
        if (cand_ref <= 0) break;
304
        // we need to continue even if cand_ref == ref.ref[0], since
305
        // the candidate could have been added as a globalmv variant,
306
        // which changes the value
307
        // FIXME if scan_{row,col}() returned a mask for the nearest
308
        // edge, we could skip the appropriate ones here
309
310
911k
        mv cand_mv = cand_b->mv.mv[n];
311
911k
        if (sign ^ sign_bias[cand_ref - 1]) {
312
8.63k
            cand_mv.y = -cand_mv.y;
313
8.63k
            cand_mv.x = -cand_mv.x;
314
8.63k
        }
315
316
911k
        int m;
317
911k
        const int last = *cnt;
318
998k
        for (m = 0; m < last; m++)
319
855k
            if (cand_mv.n == mvstack[m].mv.mv[0].n)
320
768k
                break;
321
911k
        if (m == last) {
322
143k
            mvstack[m].mv.mv[0] = cand_mv;
323
143k
            mvstack[m].weight = 2; // "minimal"
324
143k
            *cnt = last + 1;
325
143k
        }
326
911k
    }
327
908k
}
328
329
/*
330
 * refmvs_frame allocates memory for one sbrow (32 blocks high, whole frame
331
 * wide) of 4x4-resolution refmvs_block entries for spatial MV referencing.
332
 * mvrefs_tile[] keeps a list of 35 (32 + 3 above) pointers into this memory,
333
 * and each sbrow, the bottom entries (y=27/29/31) are exchanged with the top
334
 * (-5/-3/-1) pointers by calling dav1d_refmvs_tile_sbrow_init() at the start
335
 * of each tile/sbrow.
336
 *
337
 * For temporal MV referencing, we call dav1d_refmvs_save_tmvs() at the end of
338
 * each tile/sbrow (when tile column threading is enabled), or at the start of
339
 * each interleaved sbrow (i.e. once for all tile columns together, when tile
340
 * column threading is disabled). This will copy the 4x4-resolution spatial MVs
341
 * into 8x8-resolution refmvs_temporal_block structures. Then, for subsequent
342
 * frames, at the start of each tile/sbrow (when tile column threading is
343
 * enabled) or at the start of each interleaved sbrow (when tile column
344
 * threading is disabled), we call load_tmvs(), which will project the MVs to
345
 * their respective position in the current frame.
346
 */
347
348
void dav1d_refmvs_find(const refmvs_tile *const rt,
349
                       refmvs_candidate mvstack[8], int *const cnt,
350
                       int *const ctx,
351
                       const union refmvs_refpair ref, const enum BlockSize bs,
352
                       const enum EdgeFlags edge_flags,
353
                       const int by4, const int bx4)
354
1.75M
{
355
1.75M
    const refmvs_frame *const rf = rt->rf;
356
1.75M
    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
357
1.75M
    const int bw4 = b_dim[0], w4 = imin(imin(bw4, 16), rt->tile_col.end - bx4);
358
1.75M
    const int bh4 = b_dim[1], h4 = imin(imin(bh4, 16), rt->tile_row.end - by4);
359
1.75M
    mv gmv[2], tgmv[2];
360
361
1.75M
    *cnt = 0;
362
1.75M
    assert(ref.ref[0] >=  0 && ref.ref[0] <= 8 &&
363
1.75M
           ref.ref[1] >= -1 && ref.ref[1] <= 8);
364
1.75M
    if (ref.ref[0] > 0) {
365
1.03M
        tgmv[0] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[0] - 1],
366
1.03M
                             bx4, by4, bw4, bh4, rf->frm_hdr);
367
1.03M
        gmv[0] = rf->frm_hdr->gmv[ref.ref[0] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
368
773k
                 tgmv[0] : (mv) { .n = INVALID_MV };
369
1.03M
    } else {
370
717k
        tgmv[0] = (mv) { .n = 0 };
371
717k
        gmv[0] = (mv) { .n = INVALID_MV };
372
717k
    }
373
1.75M
    if (ref.ref[1] > 0) {
374
171k
        tgmv[1] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[1] - 1],
375
171k
                             bx4, by4, bw4, bh4, rf->frm_hdr);
376
171k
        gmv[1] = rf->frm_hdr->gmv[ref.ref[1] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
377
112k
                 tgmv[1] : (mv) { .n = INVALID_MV };
378
171k
    }
379
380
    // top
381
1.75M
    int have_newmv = 0, have_col_mvs = 0, have_row_mvs = 0;
382
1.75M
    unsigned max_rows = 0, n_rows = ~0;
383
1.75M
    const refmvs_block *b_top;
384
1.75M
    if (by4 > rt->tile_row.start) {
385
1.22M
        max_rows = imin((by4 - rt->tile_row.start + 1) >> 1, 2 + (bh4 > 1));
386
1.22M
        b_top = &rt->r[(by4 & 31) - 1 + 5][bx4];
387
1.22M
        n_rows = scan_row(mvstack, cnt, ref, gmv, b_top,
388
1.22M
                          bw4, w4, max_rows, bw4 >= 16 ? 4 : 1,
389
1.22M
                          &have_newmv, &have_row_mvs);
390
1.22M
    }
391
392
    // left
393
1.75M
    unsigned max_cols = 0, n_cols = ~0U;
394
1.75M
    refmvs_block *const *b_left;
395
1.75M
    if (bx4 > rt->tile_col.start) {
396
1.44M
        max_cols = imin((bx4 - rt->tile_col.start + 1) >> 1, 2 + (bw4 > 1));
397
1.44M
        b_left = &rt->r[(by4 & 31) + 5];
398
1.44M
        n_cols = scan_col(mvstack, cnt, ref, gmv, b_left,
399
1.44M
                          bh4, h4, bx4 - 1, max_cols, bh4 >= 16 ? 4 : 1,
400
1.44M
                          &have_newmv, &have_col_mvs);
401
1.44M
    }
402
403
    // top/right
404
1.75M
    if (n_rows != ~0U && edge_flags & EDGE_I444_TOP_HAS_RIGHT &&
405
725k
        imax(bw4, bh4) <= 16 && bw4 + bx4 < rt->tile_col.end)
406
600k
    {
407
600k
        add_spatial_candidate(mvstack, cnt, 4, &b_top[bw4], ref, gmv,
408
600k
                              &have_newmv, &have_row_mvs);
409
600k
    }
410
411
1.75M
    const int nearest_match = have_col_mvs + have_row_mvs;
412
1.75M
    const int nearest_cnt = *cnt;
413
3.73M
    for (int n = 0; n < nearest_cnt; n++)
414
1.98M
        mvstack[n].weight += 640;
415
416
    // temporal
417
1.75M
    int globalmv_ctx = rf->frm_hdr->use_ref_frame_mvs;
418
1.75M
    if (rf->use_ref_frame_mvs) {
419
37.6k
        const ptrdiff_t stride = rf->rp_stride;
420
37.6k
        const int by8 = by4 >> 1, bx8 = bx4 >> 1;
421
37.6k
        const refmvs_temporal_block *const rbi = &rt->rp_proj[(by8 & 15) * stride + bx8];
422
37.6k
        const refmvs_temporal_block *rb = rbi;
423
37.6k
        const int step_h = bw4 >= 16 ? 2 : 1, step_v = bh4 >= 16 ? 2 : 1;
424
37.6k
        const int w8 = imin((w4 + 1) >> 1, 8), h8 = imin((h4 + 1) >> 1, 8);
425
113k
        for (int y = 0; y < h8; y += step_v) {
426
193k
            for (int x = 0; x < w8; x+= step_h) {
427
117k
                add_temporal_candidate(rf, mvstack, cnt, &rb[x], ref,
428
117k
                                       !(x | y) ? &globalmv_ctx : NULL, tgmv);
429
117k
            }
430
76.2k
            rb += stride * step_v;
431
76.2k
        }
432
37.6k
        if (imin(bw4, bh4) >= 2 && imax(bw4, bh4) < 16) {
433
21.7k
            const int bh8 = bh4 >> 1, bw8 = bw4 >> 1;
434
21.7k
            rb = &rbi[bh8 * stride];
435
21.7k
            const int has_bottom = by8 + bh8 < imin(rt->tile_row.end >> 1,
436
21.7k
                                                    (by8 & ~7) + 8);
437
21.7k
            if (has_bottom && bx8 - 1 >= imax(rt->tile_col.start >> 1, bx8 & ~7)) {
438
5.81k
                add_temporal_candidate(rf, mvstack, cnt, &rb[-1], ref,
439
5.81k
                                       NULL, NULL);
440
5.81k
            }
441
21.7k
            if (bx8 + bw8 < imin(rt->tile_col.end >> 1, (bx8 & ~7) + 8)) {
442
8.24k
                if (has_bottom) {
443
5.35k
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8], ref,
444
5.35k
                                           NULL, NULL);
445
5.35k
                }
446
8.24k
                if (by8 + bh8 - 1 < imin(rt->tile_row.end >> 1, (by8 & ~7) + 8)) {
447
7.92k
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8 - stride],
448
7.92k
                                           ref, NULL, NULL);
449
7.92k
                }
450
8.24k
            }
451
21.7k
        }
452
37.6k
    }
453
1.75M
    assert(*cnt <= 8);
454
455
    // top/left (which, confusingly, is part of "secondary" references)
456
1.75M
    int have_dummy_newmv_match;
457
1.75M
    if ((n_rows | n_cols) != ~0U) {
458
1.06M
        add_spatial_candidate(mvstack, cnt, 4, &b_top[-1], ref, gmv,
459
1.06M
                              &have_dummy_newmv_match, &have_row_mvs);
460
1.06M
    }
461
462
    // "secondary" (non-direct neighbour) top & left edges
463
    // what is different about secondary is that everything is now in 8x8 resolution
464
5.26M
    for (int n = 2; n <= 3; n++) {
465
3.51M
        if ((unsigned) n > n_rows && (unsigned) n <= max_rows) {
466
1.09M
            n_rows += scan_row(mvstack, cnt, ref, gmv,
467
1.09M
                               &rt->r[(((by4 & 31) - 2 * n + 1) | 1) + 5][bx4 | 1],
468
1.09M
                               bw4, w4, 1 + max_rows - n, bw4 >= 16 ? 4 : 2,
469
1.09M
                               &have_dummy_newmv_match, &have_row_mvs);
470
1.09M
        }
471
472
3.51M
        if ((unsigned) n > n_cols && (unsigned) n <= max_cols) {
473
1.61M
            n_cols += scan_col(mvstack, cnt, ref, gmv, &rt->r[((by4 & 31) | 1) + 5],
474
1.61M
                               bh4, h4, (bx4 - n * 2 + 1) | 1,
475
1.61M
                               1 + max_cols - n, bh4 >= 16 ? 4 : 2,
476
1.61M
                               &have_dummy_newmv_match, &have_col_mvs);
477
1.61M
        }
478
3.51M
    }
479
1.75M
    assert(*cnt <= 8);
480
481
1.75M
    const int ref_match_count = have_col_mvs + have_row_mvs;
482
483
    // context build-up
484
1.75M
    int refmv_ctx, newmv_ctx;
485
1.75M
    switch (nearest_match) {
486
385k
    case 0:
487
385k
        refmv_ctx = imin(2, ref_match_count);
488
385k
        newmv_ctx = ref_match_count > 0;
489
385k
        break;
490
670k
    case 1:
491
670k
        refmv_ctx = imin(ref_match_count * 3, 4);
492
670k
        newmv_ctx = 3 - have_newmv;
493
670k
        break;
494
704k
    case 2:
495
704k
        refmv_ctx = 5;
496
704k
        newmv_ctx = 5 - have_newmv;
497
704k
        break;
498
1.75M
    }
499
500
    // sorting (nearest, then "secondary")
501
1.75M
    int len = nearest_cnt;
502
3.42M
    while (len) {
503
1.66M
        int last = 0;
504
2.39M
        for (int n = 1; n < len; n++) {
505
730k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
506
394k
#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
507
312k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
508
312k
                last = n;
509
312k
            }
510
730k
        }
511
1.66M
        len = last;
512
1.66M
    }
513
1.75M
    len = *cnt;
514
2.62M
    while (len > nearest_cnt) {
515
862k
        int last = nearest_cnt;
516
1.31M
        for (int n = nearest_cnt + 1; n < len; n++) {
517
451k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
518
82.3k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
519
82.3k
#undef EXCHANGE
520
82.3k
                last = n;
521
82.3k
            }
522
451k
        }
523
862k
        len = last;
524
862k
    }
525
526
1.75M
    if (ref.ref[1] > 0) {
527
171k
        if (*cnt < 2) {
528
131k
            const int sign0 = rf->sign_bias[ref.ref[0] - 1];
529
131k
            const int sign1 = rf->sign_bias[ref.ref[1] - 1];
530
131k
            const int sz4 = imin(w4, h4);
531
131k
            refmvs_candidate *const same = &mvstack[*cnt];
532
131k
            int same_count[4] = { 0 };
533
534
            // non-self references in top
535
153k
            if (n_rows != ~0U) for (int x = 0; x < sz4;) {
536
79.7k
                const refmvs_block *const cand_b = &b_top[x];
537
79.7k
                add_compound_extended_candidate(same, same_count, cand_b,
538
79.7k
                                                sign0, sign1, ref, rf->sign_bias);
539
79.7k
                x += dav1d_block_dimensions[cand_b->bs][0];
540
79.7k
            }
541
542
            // non-self references in left
543
161k
            if (n_cols != ~0U) for (int y = 0; y < sz4;) {
544
85.0k
                const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
545
85.0k
                add_compound_extended_candidate(same, same_count, cand_b,
546
85.0k
                                                sign0, sign1, ref, rf->sign_bias);
547
85.0k
                y += dav1d_block_dimensions[cand_b->bs][1];
548
85.0k
            }
549
550
131k
            refmvs_candidate *const diff = &same[2];
551
131k
            const int *const diff_count = &same_count[2];
552
553
            // merge together
554
394k
            for (int n = 0; n < 2; n++) {
555
262k
                int m = same_count[n];
556
557
262k
                if (m >= 2) continue;
558
559
220k
                const int l = diff_count[n];
560
220k
                if (l) {
561
131k
                    same[m].mv.mv[n] = diff[0].mv.mv[n];
562
131k
                    if (++m == 2) continue;
563
49.1k
                    if (l == 2) {
564
34.8k
                        same[1].mv.mv[n] = diff[1].mv.mv[n];
565
34.8k
                        continue;
566
34.8k
                    }
567
49.1k
                }
568
183k
                do {
569
183k
                    same[m].mv.mv[n] = tgmv[n];
570
183k
                } while (++m < 2);
571
103k
            }
572
573
            // if the first extended was the same as the non-extended one,
574
            // then replace it with the second extended one
575
131k
            int n = *cnt;
576
131k
            if (n == 1 && mvstack[0].mv.n == same[0].mv.n)
577
35.5k
                mvstack[1].mv = mvstack[2].mv;
578
214k
            do {
579
214k
                mvstack[n].weight = 2;
580
214k
            } while (++n < 2);
581
131k
            *cnt = 2;
582
131k
        }
583
584
        // clamping
585
171k
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
586
171k
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
587
171k
        const int top = -(by4 + bh4 + 4) * 4 * 8;
588
171k
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
589
590
171k
        const int n_refmvs = *cnt;
591
171k
        int n = 0;
592
372k
        do {
593
372k
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
594
372k
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
595
372k
            mvstack[n].mv.mv[1].x = iclip(mvstack[n].mv.mv[1].x, left, right);
596
372k
            mvstack[n].mv.mv[1].y = iclip(mvstack[n].mv.mv[1].y, top, bottom);
597
372k
        } while (++n < n_refmvs);
598
599
171k
        switch (refmv_ctx >> 1) {
600
90.8k
        case 0:
601
90.8k
            *ctx = imin(newmv_ctx, 1);
602
90.8k
            break;
603
51.0k
        case 1:
604
51.0k
            *ctx = 1 + imin(newmv_ctx, 3);
605
51.0k
            break;
606
29.4k
        case 2:
607
29.4k
            *ctx = iclip(3 + newmv_ctx, 4, 7);
608
29.4k
            break;
609
171k
        }
610
611
171k
        return;
612
1.58M
    } else if (*cnt < 2 && ref.ref[0] > 0) {
613
645k
        const int sign = rf->sign_bias[ref.ref[0] - 1];
614
645k
        const int sz4 = imin(w4, h4);
615
616
        // non-self references in top
617
901k
        if (n_rows != ~0U) for (int x = 0; x < sz4 && *cnt < 2;) {
618
462k
            const refmvs_block *const cand_b = &b_top[x];
619
462k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
620
462k
            x += dav1d_block_dimensions[cand_b->bs][0];
621
462k
        }
622
623
        // non-self references in left
624
893k
        if (n_cols != ~0U) for (int y = 0; y < sz4 && *cnt < 2;) {
625
446k
            const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
626
446k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
627
446k
            y += dav1d_block_dimensions[cand_b->bs][1];
628
446k
        }
629
645k
    }
630
1.58M
    assert(*cnt <= 8);
631
632
    // clamping
633
1.58M
    int n_refmvs = *cnt;
634
1.58M
    if (n_refmvs) {
635
1.40M
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
636
1.40M
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
637
1.40M
        const int top = -(by4 + bh4 + 4) * 4 * 8;
638
1.40M
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
639
640
1.40M
        int n = 0;
641
3.18M
        do {
642
3.18M
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
643
3.18M
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
644
3.18M
        } while (++n < n_refmvs);
645
1.40M
    }
646
647
2.48M
    for (int n = *cnt; n < 2; n++)
648
892k
        mvstack[n].mv.mv[0] = tgmv[0];
649
650
1.58M
    *ctx = (refmv_ctx << 4) | (globalmv_ctx << 3) | newmv_ctx;
651
1.58M
}
652
653
void dav1d_refmvs_tile_sbrow_init(refmvs_tile *const rt, const refmvs_frame *const rf,
654
                                  const int tile_col_start4, const int tile_col_end4,
655
                                  const int tile_row_start4, const int tile_row_end4,
656
                                  const int sby, int tile_row_idx, const int pass)
657
484k
{
658
484k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
659
484k
    rt->rp_proj = &rf->rp_proj[16 * rf->rp_stride * tile_row_idx];
660
484k
    const ptrdiff_t r_stride = rf->rp_stride * 2;
661
484k
    const ptrdiff_t pass_off = (rf->n_frame_threads > 1 && pass == 2) ?
662
257k
        35 * 2 * rf->n_blocks : 0;
663
484k
    refmvs_block *r = &rf->r[35 * r_stride * tile_row_idx + pass_off];
664
484k
    const int sbsz = rf->sbsz;
665
484k
    const int off = (sbsz * sby) & 16;
666
10.4M
    for (int i = 0; i < sbsz; i++, r += r_stride)
667
9.92M
        rt->r[off + 5 + i] = r;
668
484k
    rt->r[off + 0] = r;
669
484k
    r += r_stride;
670
484k
    rt->r[off + 1] = NULL;
671
484k
    rt->r[off + 2] = r;
672
484k
    r += r_stride;
673
484k
    rt->r[off + 3] = NULL;
674
484k
    rt->r[off + 4] = r;
675
484k
    if (sby & 1) {
676
273k
#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
677
91.0k
        EXCHANGE(rt->r[off + 0], rt->r[off + sbsz + 0]);
678
91.0k
        EXCHANGE(rt->r[off + 2], rt->r[off + sbsz + 2]);
679
91.0k
        EXCHANGE(rt->r[off + 4], rt->r[off + sbsz + 4]);
680
91.0k
#undef EXCHANGE
681
91.0k
    }
682
683
484k
    rt->rf = rf;
684
484k
    rt->tile_row.start = tile_row_start4;
685
484k
    rt->tile_row.end = imin(tile_row_end4, rf->ih4);
686
484k
    rt->tile_col.start = tile_col_start4;
687
484k
    rt->tile_col.end = imin(tile_col_end4, rf->iw4);
688
484k
}
689
690
static void load_tmvs_c(const refmvs_frame *const rf, int tile_row_idx,
691
                        const int col_start8, const int col_end8,
692
                        const int row_start8, int row_end8)
693
20.8k
{
694
20.8k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
695
20.8k
    assert(row_start8 >= 0);
696
20.8k
    assert((unsigned) (row_end8 - row_start8) <= 16U);
697
20.8k
    row_end8 = imin(row_end8, rf->ih8);
698
20.8k
    const int col_start8i = imax(col_start8 - 8, 0);
699
20.8k
    const int col_end8i = imin(col_end8 + 8, rf->iw8);
700
701
20.8k
    const ptrdiff_t stride = rf->rp_stride;
702
20.8k
    refmvs_temporal_block *rp_proj =
703
20.8k
        &rf->rp_proj[16 * stride * tile_row_idx + (row_start8 & 15) * stride];
704
139k
    for (int y = row_start8; y < row_end8; y++) {
705
474k
        for (int x = col_start8; x < col_end8; x++)
706
356k
            rp_proj[x].mv.n = INVALID_MV;
707
118k
        rp_proj += stride;
708
118k
    }
709
710
20.8k
    rp_proj = &rf->rp_proj[16 * stride * tile_row_idx];
711
42.0k
    for (int n = 0; n < rf->n_mfmvs; n++) {
712
21.1k
        const int ref2cur = rf->mfmv_ref2cur[n];
713
21.1k
        if (ref2cur == INVALID_REF2CUR) continue;
714
715
18.9k
        const int ref = rf->mfmv_ref[n];
716
18.9k
        const int ref_sign = ref - 4;
717
18.9k
        const refmvs_temporal_block *r = &rf->rp_ref[ref][row_start8 * stride];
718
111k
        for (int y = row_start8; y < row_end8; y++) {
719
93.0k
            const int y_sb_align = y & ~7;
720
93.0k
            const int y_proj_start = imax(y_sb_align, row_start8);
721
93.0k
            const int y_proj_end = imin(y_sb_align + 8, row_end8);
722
240k
            for (int x = col_start8i; x < col_end8i; x++) {
723
147k
                const refmvs_temporal_block *rb = &r[x];
724
147k
                const int b_ref = rb->ref;
725
147k
                if (!b_ref) continue;
726
54.5k
                const int ref2ref = rf->mfmv_ref2ref[n][b_ref - 1];
727
54.5k
                if (!ref2ref) continue;
728
42.4k
                const mv b_mv = rb->mv;
729
42.4k
                const mv offset = mv_projection(b_mv, ref2cur, ref2ref);
730
42.4k
                int pos_x = x + apply_sign(abs(offset.x) >> 6,
731
42.4k
                                           offset.x ^ ref_sign);
732
42.4k
                const int pos_y = y + apply_sign(abs(offset.y) >> 6,
733
42.4k
                                                 offset.y ^ ref_sign);
734
42.4k
                if (pos_y >= y_proj_start && pos_y < y_proj_end) {
735
38.7k
                    const ptrdiff_t pos = (pos_y & 15) * stride;
736
96.4k
                    for (;;) {
737
96.4k
                        const int x_sb_align = x & ~7;
738
96.4k
                        if (pos_x >= imax(x_sb_align - 8, col_start8) &&
739
95.9k
                            pos_x < imin(x_sb_align + 16, col_end8))
740
93.9k
                        {
741
93.9k
                            rp_proj[pos + pos_x].mv = rb->mv;
742
93.9k
                            rp_proj[pos + pos_x].ref = ref2ref;
743
93.9k
                        }
744
96.4k
                        if (++x >= col_end8i) break;
745
67.6k
                        rb++;
746
67.6k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
747
57.7k
                        pos_x++;
748
57.7k
                    }
749
38.7k
                } else {
750
8.22k
                    for (;;) {
751
8.22k
                        if (++x >= col_end8i) break;
752
6.02k
                        rb++;
753
6.02k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
754
6.02k
                    }
755
3.70k
                }
756
42.4k
                x--;
757
42.4k
            }
758
93.0k
            r += stride;
759
93.0k
        }
760
18.9k
    }
761
20.8k
}
762
763
static void save_tmvs_c(refmvs_temporal_block *rp, const ptrdiff_t stride,
764
                        refmvs_block *const *const rr,
765
                        const uint8_t *const ref_sign,
766
                        const int col_end8, const int row_end8,
767
                        const int col_start8, const int row_start8)
768
62.7k
{
769
422k
    for (int y = row_start8; y < row_end8; y++) {
770
360k
        const refmvs_block *const b = rr[(y & 15) * 2];
771
772
830k
        for (int x = col_start8; x < col_end8;) {
773
470k
            const refmvs_block *const cand_b = &b[x * 2 + 1];
774
470k
            const int bw8 = (dav1d_block_dimensions[cand_b->bs][0] + 1) >> 1;
775
776
470k
            if (cand_b->ref.ref[1] > 0 && ref_sign[cand_b->ref.ref[1] - 1] &&
777
54.8k
                (abs(cand_b->mv.mv[1].y) | abs(cand_b->mv.mv[1].x)) < 4096)
778
47.3k
            {
779
47.3k
                const refmvs_temporal_block tmv = {
780
47.3k
                    .mv = cand_b->mv.mv[1],
781
47.3k
                    .ref = cand_b->ref.ref[1],
782
47.3k
                };
783
180k
                for (int n = 0; n < bw8; n++, x++)
784
133k
                    rp[x] = tmv;
785
422k
            } else if (cand_b->ref.ref[0] > 0 && ref_sign[cand_b->ref.ref[0] - 1] &&
786
139k
                       (abs(cand_b->mv.mv[0].y) | abs(cand_b->mv.mv[0].x)) < 4096)
787
132k
            {
788
132k
                const refmvs_temporal_block tmv = {
789
132k
                    .mv = cand_b->mv.mv[0],
790
132k
                    .ref = cand_b->ref.ref[0],
791
132k
                };
792
540k
                for (int n = 0; n < bw8; n++, x++)
793
407k
                    rp[x] = tmv;
794
289k
            } else {
795
289k
                const refmvs_temporal_block tmv = { .mv = { .n = 0 }, .ref = 0 };
796
1.33M
                for (int n = 0; n < bw8; n++, x++)
797
1.04M
                    rp[x] = tmv;
798
289k
            }
799
470k
        }
800
360k
        rp += stride;
801
360k
    }
802
62.7k
}
803
804
int dav1d_refmvs_init_frame(refmvs_frame *const rf,
805
                            const Dav1dSequenceHeader *const seq_hdr,
806
                            const Dav1dFrameHeader *const frm_hdr,
807
                            const uint8_t ref_poc[7],
808
                            refmvs_temporal_block *const rp,
809
                            const uint8_t ref_ref_poc[7][7],
810
                            /*const*/ refmvs_temporal_block *const rp_ref[7],
811
                            const int n_tile_threads, const int n_frame_threads)
812
183k
{
813
183k
    const int rp_stride = ((frm_hdr->width[0] + 127) & ~127) >> 3;
814
18.4E
    const int n_tile_rows = n_tile_threads > 1 ? frm_hdr->tiling.rows : 1;
815
183k
    const int n_blocks = rp_stride * n_tile_rows;
816
817
183k
    rf->sbsz = 16 << seq_hdr->sb128;
818
183k
    rf->frm_hdr = frm_hdr;
819
183k
    rf->iw8 = (frm_hdr->width[0] + 7) >> 3;
820
183k
    rf->ih8 = (frm_hdr->height + 7) >> 3;
821
183k
    rf->iw4 = rf->iw8 << 1;
822
183k
    rf->ih4 = rf->ih8 << 1;
823
183k
    rf->rp = rp;
824
183k
    rf->rp_stride = rp_stride;
825
183k
    rf->n_tile_threads = n_tile_threads;
826
183k
    rf->n_frame_threads = n_frame_threads;
827
828
183k
    if (n_blocks != rf->n_blocks) {
829
61.6k
        const size_t r_sz = sizeof(*rf->r) * 35 * 2 * n_blocks * (1 + (n_frame_threads > 1));
830
61.6k
        const size_t rp_proj_sz = sizeof(*rf->rp_proj) * 16 * n_blocks;
831
        /* Note that sizeof(*rf->r) == 12, but it's accessed using 16-byte unaligned
832
         * loads in save_tmvs() asm which can overread 4 bytes into rp_proj. */
833
61.6k
        dav1d_free_aligned(rf->r);
834
61.6k
        rf->r = dav1d_alloc_aligned(ALLOC_REFMVS, r_sz + rp_proj_sz, 64);
835
61.6k
        if (!rf->r) {
836
0
            rf->n_blocks = 0;
837
0
            return DAV1D_ERR(ENOMEM);
838
0
        }
839
840
61.6k
        rf->rp_proj = (refmvs_temporal_block*)((uintptr_t)rf->r + r_sz);
841
61.6k
        rf->n_blocks = n_blocks;
842
61.6k
    }
843
844
183k
    const int poc = frm_hdr->frame_offset;
845
1.46M
    for (int i = 0; i < 7; i++) {
846
1.28M
        const int poc_diff = get_poc_diff(seq_hdr->order_hint_n_bits,
847
1.28M
                                          ref_poc[i], poc);
848
1.28M
        rf->sign_bias[i] = poc_diff > 0;
849
1.28M
        rf->mfmv_sign[i] = poc_diff < 0;
850
1.28M
        rf->pocdiff[i] = iclip(get_poc_diff(seq_hdr->order_hint_n_bits,
851
1.28M
                                            poc, ref_poc[i]), -31, 31);
852
1.28M
    }
853
854
    // temporal MV setup
855
183k
    rf->n_mfmvs = 0;
856
183k
    rf->rp_ref = rp_ref;
857
183k
    if (frm_hdr->use_ref_frame_mvs && seq_hdr->order_hint_n_bits) {
858
17.5k
        int total = 2;
859
17.5k
        if (rp_ref[0] && ref_ref_poc[0][6] != ref_poc[3] /* alt-of-last != gold */) {
860
4.18k
            rf->mfmv_ref[rf->n_mfmvs++] = 0; // last
861
4.18k
            total = 3;
862
4.18k
        }
863
17.5k
        if (rp_ref[4] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[4],
864
12.4k
                                      frm_hdr->frame_offset) > 0)
865
1.13k
        {
866
1.13k
            rf->mfmv_ref[rf->n_mfmvs++] = 4; // bwd
867
1.13k
        }
868
17.5k
        if (rp_ref[5] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[5],
869
11.5k
                                      frm_hdr->frame_offset) > 0)
870
751
        {
871
751
            rf->mfmv_ref[rf->n_mfmvs++] = 5; // altref2
872
751
        }
873
17.5k
        if (rf->n_mfmvs < total && rp_ref[6] &&
874
7.45k
            get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[6],
875
7.45k
                         frm_hdr->frame_offset) > 0)
876
2.11k
        {
877
2.11k
            rf->mfmv_ref[rf->n_mfmvs++] = 6; // altref
878
2.11k
        }
879
17.5k
        if (rf->n_mfmvs < total && rp_ref[1])
880
11.1k
            rf->mfmv_ref[rf->n_mfmvs++] = 1; // last2
881
882
36.8k
        for (int n = 0; n < rf->n_mfmvs; n++) {
883
19.3k
            const int rpoc = ref_poc[rf->mfmv_ref[n]];
884
19.3k
            const int diff1 = get_poc_diff(seq_hdr->order_hint_n_bits,
885
19.3k
                                           rpoc, frm_hdr->frame_offset);
886
19.3k
            if (abs(diff1) > 31) {
887
2.15k
                rf->mfmv_ref2cur[n] = INVALID_REF2CUR;
888
17.1k
            } else {
889
17.1k
                rf->mfmv_ref2cur[n] = rf->mfmv_ref[n] < 4 ? -diff1 : diff1;
890
137k
                for (int m = 0; m < 7; m++) {
891
120k
                    const int rrpoc = ref_ref_poc[rf->mfmv_ref[n]][m];
892
120k
                    const int diff2 = get_poc_diff(seq_hdr->order_hint_n_bits,
893
120k
                                                   rpoc, rrpoc);
894
                    // unsigned comparison also catches the < 0 case
895
120k
                    rf->mfmv_ref2ref[n][m] = (unsigned) diff2 > 31U ? 0 : diff2;
896
120k
                }
897
17.1k
            }
898
19.3k
        }
899
17.5k
    }
900
183k
    rf->use_ref_frame_mvs = rf->n_mfmvs > 0;
901
902
183k
    return 0;
903
183k
}
904
905
static void splat_mv_c(refmvs_block **rr, const refmvs_block *const rmv,
906
                       const int bx4, const int bw4, int bh4)
907
3.04M
{
908
12.4M
    do {
909
12.4M
        refmvs_block *const r = *rr++ + bx4;
910
118M
        for (int x = 0; x < bw4; x++)
911
106M
            r[x] = *rmv;
912
12.4M
    } while (--bh4);
913
3.04M
}
914
915
#if HAVE_ASM
916
#if ARCH_AARCH64 || ARCH_ARM
917
#include "src/arm/refmvs.h"
918
#elif ARCH_LOONGARCH64
919
#include "src/loongarch/refmvs.h"
920
#elif ARCH_X86
921
#include "src/x86/refmvs.h"
922
#endif
923
#endif
924
925
COLD void dav1d_refmvs_dsp_init(Dav1dRefmvsDSPContext *const c)
926
84.9k
{
927
84.9k
    c->load_tmvs = load_tmvs_c;
928
84.9k
    c->save_tmvs = save_tmvs_c;
929
84.9k
    c->splat_mv = splat_mv_c;
930
931
#if HAVE_ASM
932
#if ARCH_AARCH64 || ARCH_ARM
933
    refmvs_dsp_init_arm(c);
934
#elif ARCH_LOONGARCH64
935
    refmvs_dsp_init_loongarch(c);
936
#elif ARCH_X86
937
    refmvs_dsp_init_x86(c);
938
#endif
939
#endif
940
84.9k
}