Coverage Report

Created: 2026-09-13 06:34

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/dav1d/src/refmvs.c
Line
Count
Source
1
/*
2
 * Copyright © 2020, VideoLAN and dav1d authors
3
 * Copyright © 2020, Two Orioles, LLC
4
 * All rights reserved.
5
 *
6
 * Redistribution and use in source and binary forms, with or without
7
 * modification, are permitted provided that the following conditions are met:
8
 *
9
 * 1. Redistributions of source code must retain the above copyright notice, this
10
 *    list of conditions and the following disclaimer.
11
 *
12
 * 2. Redistributions in binary form must reproduce the above copyright notice,
13
 *    this list of conditions and the following disclaimer in the documentation
14
 *    and/or other materials provided with the distribution.
15
 *
16
 * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
17
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
18
 * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
19
 * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
20
 * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
21
 * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
22
 * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
23
 * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
24
 * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
25
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
26
 */
27
28
#include "config.h"
29
30
#include <limits.h>
31
#include <stdlib.h>
32
33
#include "dav1d/common.h"
34
35
#include "common/intops.h"
36
37
#include "src/env.h"
38
#include "src/mem.h"
39
#include "src/refmvs.h"
40
41
static void add_spatial_candidate(refmvs_candidate *const mvstack, int *const cnt,
42
                                  const int weight, const refmvs_block *const b,
43
                                  const union refmvs_refpair ref, const mv gmv[2],
44
                                  int *const have_newmv_match,
45
                                  int *const have_refmv_match)
46
2.53M
{
47
2.53M
    if (b->mv.mv[0].n == INVALID_MV) return; // intra block, no intrabc
48
49
2.09M
    if (ref.ref[1] == -1) {
50
2.34M
        for (int n = 0; n < 2; n++) {
51
2.06M
            if (b->ref.ref[n] == ref.ref[0]) {
52
1.42M
                const mv cand_mv = ((b->mf & 1) && gmv[0].n != INVALID_MV) ?
53
1.39M
                                   gmv[0] : b->mv.mv[n];
54
55
1.42M
                *have_refmv_match = 1;
56
1.42M
                *have_newmv_match |= b->mf >> 1;
57
58
1.42M
                const int last = *cnt;
59
2.78M
                for (int m = 0; m < last; m++)
60
1.84M
                    if (mvstack[m].mv.mv[0].n == cand_mv.n) {
61
492k
                        mvstack[m].weight += weight;
62
492k
                        return;
63
492k
                    }
64
65
936k
                if (last < 8) {
66
934k
                    mvstack[last].mv.mv[0] = cand_mv;
67
934k
                    mvstack[last].weight = weight;
68
934k
                    *cnt = last + 1;
69
934k
                }
70
936k
                return;
71
1.42M
            }
72
2.06M
        }
73
1.71M
    } else if (b->ref.pair == ref.pair) {
74
117k
        const refmvs_mvpair cand_mv = { .mv = {
75
117k
            [0] = ((b->mf & 1) && gmv[0].n != INVALID_MV) ? gmv[0] : b->mv.mv[0],
76
117k
            [1] = ((b->mf & 1) && gmv[1].n != INVALID_MV) ? gmv[1] : b->mv.mv[1],
77
117k
        }};
78
79
117k
        *have_refmv_match = 1;
80
117k
        *have_newmv_match |= b->mf >> 1;
81
82
117k
        const int last = *cnt;
83
171k
        for (int n = 0; n < last; n++)
84
99.4k
            if (mvstack[n].mv.n == cand_mv.n) {
85
44.8k
                mvstack[n].weight += weight;
86
44.8k
                return;
87
44.8k
            }
88
89
72.4k
        if (last < 8) {
90
72.4k
            mvstack[last].mv = cand_mv;
91
72.4k
            mvstack[last].weight = weight;
92
72.4k
            *cnt = last + 1;
93
72.4k
        }
94
72.4k
    }
95
2.09M
}
96
97
static int scan_row(refmvs_candidate *const mvstack, int *const cnt,
98
                    const union refmvs_refpair ref, const mv gmv[2],
99
                    const refmvs_block *b, const int bw4, const int w4,
100
                    const int max_rows, const int step,
101
                    int *const have_newmv_match, int *const have_refmv_match)
102
800k
{
103
800k
    const refmvs_block *cand_b = b;
104
800k
    const enum BlockSize first_cand_bs = cand_b->bs;
105
800k
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
106
800k
    int cand_bw4 = first_cand_b_dim[0];
107
800k
    int len = imax(step, imin(bw4, cand_bw4));
108
109
800k
    if (bw4 <= cand_bw4) {
110
        // FIXME weight can be higher for odd blocks (bx4 & 1), but then the
111
        // position of the first block has to be odd already, i.e. not just
112
        // for row_offset=-3/-5
113
        // FIXME why can this not be cand_bw4?
114
699k
        const int weight = bw4 == 1 ? 2 :
115
699k
                           imax(2, imin(2 * max_rows, first_cand_b_dim[1]));
116
699k
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
117
699k
                              have_newmv_match, have_refmv_match);
118
699k
        return weight >> 1;
119
699k
    }
120
121
202k
    for (int x = 0;;) {
122
        // FIXME if we overhang above, we could fill a bitmask so we don't have
123
        // to repeat the add_spatial_candidate() for the next row, but just increase
124
        // the weight here
125
202k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
126
202k
                              have_newmv_match, have_refmv_match);
127
202k
        x += len;
128
202k
        if (x >= w4) return 1;
129
101k
        cand_b = &b[x];
130
101k
        cand_bw4 = dav1d_block_dimensions[cand_b->bs][0];
131
101k
        assert(cand_bw4 < bw4);
132
101k
        len = imax(step, cand_bw4);
133
101k
    }
134
101k
}
135
136
static int scan_col(refmvs_candidate *const mvstack, int *const cnt,
137
                    const union refmvs_refpair ref, const mv gmv[2],
138
                    /*const*/ refmvs_block *const *b, const int bh4, const int h4,
139
                    const int bx4, const int max_cols, const int step,
140
                    int *const have_newmv_match, int *const have_refmv_match)
141
951k
{
142
951k
    const refmvs_block *cand_b = &b[0][bx4];
143
951k
    const enum BlockSize first_cand_bs = cand_b->bs;
144
951k
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
145
951k
    int cand_bh4 = first_cand_b_dim[1];
146
951k
    int len = imax(step, imin(bh4, cand_bh4));
147
148
951k
    if (bh4 <= cand_bh4) {
149
        // FIXME weight can be higher for odd blocks (by4 & 1), but then the
150
        // position of the first block has to be odd already, i.e. not just
151
        // for col_offset=-3/-5
152
        // FIXME why can this not be cand_bh4?
153
825k
        const int weight = bh4 == 1 ? 2 :
154
825k
                           imax(2, imin(2 * max_cols, first_cand_b_dim[0]));
155
825k
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
156
825k
                            have_newmv_match, have_refmv_match);
157
825k
        return weight >> 1;
158
825k
    }
159
160
249k
    for (int y = 0;;) {
161
        // FIXME if we overhang above, we could fill a bitmask so we don't have
162
        // to repeat the add_spatial_candidate() for the next row, but just increase
163
        // the weight here
164
249k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
165
249k
                              have_newmv_match, have_refmv_match);
166
249k
        y += len;
167
249k
        if (y >= h4) return 1;
168
122k
        cand_b = &b[y][bx4];
169
122k
        cand_bh4 = dav1d_block_dimensions[cand_b->bs][1];
170
122k
        assert(cand_bh4 < bh4);
171
122k
        len = imax(step, cand_bh4);
172
122k
    }
173
126k
}
174
175
102k
static inline union mv mv_projection(const union mv mv, const int num, const int den) {
176
102k
    static const uint16_t div_mult[32] = {
177
102k
           0, 16384, 8192, 5461, 4096, 3276, 2730, 2340,
178
102k
        2048,  1820, 1638, 1489, 1365, 1260, 1170, 1092,
179
102k
        1024,   963,  910,  862,  819,  780,  744,  712,
180
102k
         682,   655,  630,  606,  585,  564,  546,  528
181
102k
    };
182
102k
    assert(den > 0 && den < 32);
183
102k
    assert(num > -32 && num < 32);
184
102k
    const int frac = num * div_mult[den];
185
102k
    const int y = mv.y * frac, x = mv.x * frac;
186
    // Round and clip according to AV1 spec section 7.9.3
187
102k
    return (union mv) { // 0x3fff == (1 << 14) - 1
188
102k
        .y = iclip((y + 8192 + (y >> 31)) >> 14, -0x3fff, 0x3fff),
189
102k
        .x = iclip((x + 8192 + (x >> 31)) >> 14, -0x3fff, 0x3fff)
190
102k
    };
191
102k
}
192
193
static void add_temporal_candidate(const refmvs_frame *const rf,
194
                                   refmvs_candidate *const mvstack, int *const cnt,
195
                                   const refmvs_temporal_block *const rb,
196
                                   const union refmvs_refpair ref, int *const globalmv_ctx,
197
                                   const union mv gmv[])
198
108k
{
199
108k
    if (rb->mv.n == INVALID_MV) return;
200
201
36.4k
    union mv mv = mv_projection(rb->mv, rf->pocdiff[ref.ref[0] - 1], rb->ref);
202
36.4k
    fix_mv_precision(rf->frm_hdr, &mv);
203
204
36.4k
    const int last = *cnt;
205
36.4k
    if (ref.ref[1] == -1) {
206
21.9k
        if (globalmv_ctx)
207
6.29k
            *globalmv_ctx = (abs(mv.x - gmv[0].x) | abs(mv.y - gmv[0].y)) >= 16;
208
209
27.4k
        for (int n = 0; n < last; n++)
210
21.7k
            if (mvstack[n].mv.mv[0].n == mv.n) {
211
16.2k
                mvstack[n].weight += 2;
212
16.2k
                return;
213
16.2k
            }
214
5.71k
        if (last < 8) {
215
5.71k
            mvstack[last].mv.mv[0] = mv;
216
5.71k
            mvstack[last].weight = 2;
217
5.71k
            *cnt = last + 1;
218
5.71k
        }
219
14.4k
    } else {
220
14.4k
        refmvs_mvpair mvp = { .mv = {
221
14.4k
            [0] = mv,
222
14.4k
            [1] = mv_projection(rb->mv, rf->pocdiff[ref.ref[1] - 1], rb->ref),
223
14.4k
        }};
224
14.4k
        fix_mv_precision(rf->frm_hdr, &mvp.mv[1]);
225
226
18.0k
        for (int n = 0; n < last; n++)
227
13.7k
            if (mvstack[n].mv.n == mvp.n) {
228
10.1k
                mvstack[n].weight += 2;
229
10.1k
                return;
230
10.1k
            }
231
4.36k
        if (last < 8) {
232
4.36k
            mvstack[last].mv = mvp;
233
4.36k
            mvstack[last].weight = 2;
234
4.36k
            *cnt = last + 1;
235
4.36k
        }
236
4.36k
    }
237
36.4k
}
238
239
static void add_compound_extended_candidate(refmvs_candidate *const same,
240
                                            int *const same_count,
241
                                            const refmvs_block *const cand_b,
242
                                            const int sign0, const int sign1,
243
                                            const union refmvs_refpair ref,
244
                                            const uint8_t *const sign_bias)
245
97.5k
{
246
97.5k
    refmvs_candidate *const diff = &same[2];
247
97.5k
    int *const diff_count = &same_count[2];
248
249
251k
    for (int n = 0; n < 2; n++) {
250
192k
        const int cand_ref = cand_b->ref.ref[n];
251
252
192k
        if (cand_ref <= 0) break;
253
254
154k
        mv cand_mv = cand_b->mv.mv[n];
255
154k
        if (cand_ref == ref.ref[0]) {
256
54.9k
            if (same_count[0] < 2)
257
53.0k
                same[same_count[0]++].mv.mv[0] = cand_mv;
258
54.9k
            if (diff_count[1] < 2) {
259
46.5k
                if (sign1 ^ sign_bias[cand_ref - 1]) {
260
2.91k
                    cand_mv.y = -cand_mv.y;
261
2.91k
                    cand_mv.x = -cand_mv.x;
262
2.91k
                }
263
46.5k
                diff[diff_count[1]++].mv.mv[1] = cand_mv;
264
46.5k
            }
265
99.4k
        } else if (cand_ref == ref.ref[1]) {
266
52.3k
            if (same_count[1] < 2)
267
51.0k
                same[same_count[1]++].mv.mv[1] = cand_mv;
268
52.3k
            if (diff_count[0] < 2) {
269
42.9k
                if (sign0 ^ sign_bias[cand_ref - 1]) {
270
2.96k
                    cand_mv.y = -cand_mv.y;
271
2.96k
                    cand_mv.x = -cand_mv.x;
272
2.96k
                }
273
42.9k
                diff[diff_count[0]++].mv.mv[0] = cand_mv;
274
42.9k
            }
275
52.3k
        } else {
276
47.1k
            mv i_cand_mv = (union mv) {
277
47.1k
                .x = -cand_mv.x,
278
47.1k
                .y = -cand_mv.y
279
47.1k
            };
280
281
47.1k
            if (diff_count[0] < 2) {
282
37.1k
                diff[diff_count[0]++].mv.mv[0] =
283
37.1k
                    sign0 ^ sign_bias[cand_ref - 1] ?
284
35.3k
                    i_cand_mv : cand_mv;
285
37.1k
            }
286
287
47.1k
            if (diff_count[1] < 2) {
288
34.3k
                diff[diff_count[1]++].mv.mv[1] =
289
34.3k
                    sign1 ^ sign_bias[cand_ref - 1] ?
290
32.7k
                    i_cand_mv : cand_mv;
291
34.3k
            }
292
47.1k
        }
293
154k
    }
294
97.5k
}
295
296
static void add_single_extended_candidate(refmvs_candidate mvstack[8], int *const cnt,
297
                                          const refmvs_block *const cand_b,
298
                                          const int sign, const uint8_t *const sign_bias)
299
176k
{
300
361k
    for (int n = 0; n < 2; n++) {
301
344k
        const int cand_ref = cand_b->ref.ref[n];
302
303
344k
        if (cand_ref <= 0) break;
304
        // we need to continue even if cand_ref == ref.ref[0], since
305
        // the candidate could have been added as a globalmv variant,
306
        // which changes the value
307
        // FIXME if scan_{row,col}() returned a mask for the nearest
308
        // edge, we could skip the appropriate ones here
309
310
184k
        mv cand_mv = cand_b->mv.mv[n];
311
184k
        if (sign ^ sign_bias[cand_ref - 1]) {
312
4.58k
            cand_mv.y = -cand_mv.y;
313
4.58k
            cand_mv.x = -cand_mv.x;
314
4.58k
        }
315
316
184k
        int m;
317
184k
        const int last = *cnt;
318
219k
        for (m = 0; m < last; m++)
319
167k
            if (cand_mv.n == mvstack[m].mv.mv[0].n)
320
132k
                break;
321
184k
        if (m == last) {
322
52.1k
            mvstack[m].mv.mv[0] = cand_mv;
323
52.1k
            mvstack[m].weight = 2; // "minimal"
324
52.1k
            *cnt = last + 1;
325
52.1k
        }
326
184k
    }
327
176k
}
328
329
/*
330
 * refmvs_frame allocates memory for one sbrow (32 blocks high, whole frame
331
 * wide) of 4x4-resolution refmvs_block entries for spatial MV referencing.
332
 * mvrefs_tile[] keeps a list of 35 (32 + 3 above) pointers into this memory,
333
 * and each sbrow, the bottom entries (y=27/29/31) are exchanged with the top
334
 * (-5/-3/-1) pointers by calling dav1d_refmvs_tile_sbrow_init() at the start
335
 * of each tile/sbrow.
336
 *
337
 * For temporal MV referencing, we call dav1d_refmvs_save_tmvs() at the end of
338
 * each tile/sbrow (when tile column threading is enabled), or at the start of
339
 * each interleaved sbrow (i.e. once for all tile columns together, when tile
340
 * column threading is disabled). This will copy the 4x4-resolution spatial MVs
341
 * into 8x8-resolution refmvs_temporal_block structures. Then, for subsequent
342
 * frames, at the start of each tile/sbrow (when tile column threading is
343
 * enabled) or at the start of each interleaved sbrow (when tile column
344
 * threading is disabled), we call load_tmvs(), which will project the MVs to
345
 * their respective position in the current frame.
346
 */
347
348
void dav1d_refmvs_find(const refmvs_tile *const rt,
349
                       refmvs_candidate mvstack[8], int *const cnt,
350
                       int *const ctx,
351
                       const union refmvs_refpair ref, const enum BlockSize bs,
352
                       const enum EdgeFlags edge_flags,
353
                       const int by4, const int bx4)
354
586k
{
355
586k
    const refmvs_frame *const rf = rt->rf;
356
586k
    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
357
586k
    const int bw4 = b_dim[0], w4 = imin(imin(bw4, 16), rt->tile_col.end - bx4);
358
586k
    const int bh4 = b_dim[1], h4 = imin(imin(bh4, 16), rt->tile_row.end - by4);
359
586k
    mv gmv[2], tgmv[2];
360
361
586k
    *cnt = 0;
362
586k
    assert(ref.ref[0] >=  0 && ref.ref[0] <= 8 &&
363
586k
           ref.ref[1] >= -1 && ref.ref[1] <= 8);
364
586k
    if (ref.ref[0] > 0) {
365
395k
        tgmv[0] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[0] - 1],
366
395k
                             bx4, by4, bw4, bh4, rf->frm_hdr);
367
395k
        gmv[0] = rf->frm_hdr->gmv[ref.ref[0] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
368
259k
                 tgmv[0] : (mv) { .n = INVALID_MV };
369
395k
    } else {
370
191k
        tgmv[0] = (mv) { .n = 0 };
371
191k
        gmv[0] = (mv) { .n = INVALID_MV };
372
191k
    }
373
586k
    if (ref.ref[1] > 0) {
374
96.8k
        tgmv[1] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[1] - 1],
375
96.8k
                             bx4, by4, bw4, bh4, rf->frm_hdr);
376
96.8k
        gmv[1] = rf->frm_hdr->gmv[ref.ref[1] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
377
72.7k
                 tgmv[1] : (mv) { .n = INVALID_MV };
378
96.8k
    }
379
380
    // top
381
586k
    int have_newmv = 0, have_col_mvs = 0, have_row_mvs = 0;
382
586k
    unsigned max_rows = 0, n_rows = ~0;
383
586k
    const refmvs_block *b_top;
384
586k
    if (by4 > rt->tile_row.start) {
385
408k
        max_rows = imin((by4 - rt->tile_row.start + 1) >> 1, 2 + (bh4 > 1));
386
408k
        b_top = &rt->r[(by4 & 31) - 1 + 5][bx4];
387
408k
        n_rows = scan_row(mvstack, cnt, ref, gmv, b_top,
388
408k
                          bw4, w4, max_rows, bw4 >= 16 ? 4 : 1,
389
408k
                          &have_newmv, &have_row_mvs);
390
408k
    }
391
392
    // left
393
586k
    unsigned max_cols = 0, n_cols = ~0U;
394
586k
    refmvs_block *const *b_left;
395
586k
    if (bx4 > rt->tile_col.start) {
396
448k
        max_cols = imin((bx4 - rt->tile_col.start + 1) >> 1, 2 + (bw4 > 1));
397
448k
        b_left = &rt->r[(by4 & 31) + 5];
398
448k
        n_cols = scan_col(mvstack, cnt, ref, gmv, b_left,
399
448k
                          bh4, h4, bx4 - 1, max_cols, bh4 >= 16 ? 4 : 1,
400
448k
                          &have_newmv, &have_col_mvs);
401
448k
    }
402
403
    // top/right
404
586k
    if (n_rows != ~0U && edge_flags & EDGE_I444_TOP_HAS_RIGHT &&
405
239k
        imax(bw4, bh4) <= 16 && bw4 + bx4 < rt->tile_col.end)
406
204k
    {
407
204k
        add_spatial_candidate(mvstack, cnt, 4, &b_top[bw4], ref, gmv,
408
204k
                              &have_newmv, &have_row_mvs);
409
204k
    }
410
411
586k
    const int nearest_match = have_col_mvs + have_row_mvs;
412
586k
    const int nearest_cnt = *cnt;
413
1.20M
    for (int n = 0; n < nearest_cnt; n++)
414
620k
        mvstack[n].weight += 640;
415
416
    // temporal
417
586k
    int globalmv_ctx = rf->frm_hdr->use_ref_frame_mvs;
418
586k
    if (rf->use_ref_frame_mvs) {
419
41.0k
        const ptrdiff_t stride = rf->rp_stride;
420
41.0k
        const int by8 = by4 >> 1, bx8 = bx4 >> 1;
421
41.0k
        const refmvs_temporal_block *const rbi = &rt->rp_proj[(by8 & 15) * stride + bx8];
422
41.0k
        const refmvs_temporal_block *rb = rbi;
423
41.0k
        const int step_h = bw4 >= 16 ? 2 : 1, step_v = bh4 >= 16 ? 2 : 1;
424
41.0k
        const int w8 = imin((w4 + 1) >> 1, 8), h8 = imin((h4 + 1) >> 1, 8);
425
129k
        for (int y = 0; y < h8; y += step_v) {
426
194k
            for (int x = 0; x < w8; x+= step_h) {
427
106k
                add_temporal_candidate(rf, mvstack, cnt, &rb[x], ref,
428
106k
                                       !(x | y) ? &globalmv_ctx : NULL, tgmv);
429
106k
            }
430
88.3k
            rb += stride * step_v;
431
88.3k
        }
432
41.0k
        if (imin(bw4, bh4) >= 2 && imax(bw4, bh4) < 16) {
433
19.4k
            const int bh8 = bh4 >> 1, bw8 = bw4 >> 1;
434
19.4k
            rb = &rbi[bh8 * stride];
435
19.4k
            const int has_bottom = by8 + bh8 < imin(rt->tile_row.end >> 1,
436
19.4k
                                                    (by8 & ~7) + 8);
437
19.4k
            if (has_bottom && bx8 - 1 >= imax(rt->tile_col.start >> 1, bx8 & ~7)) {
438
803
                add_temporal_candidate(rf, mvstack, cnt, &rb[-1], ref,
439
803
                                       NULL, NULL);
440
803
            }
441
19.4k
            if (bx8 + bw8 < imin(rt->tile_col.end >> 1, (bx8 & ~7) + 8)) {
442
1.23k
                if (has_bottom) {
443
808
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8], ref,
444
808
                                           NULL, NULL);
445
808
                }
446
1.23k
                if (by8 + bh8 - 1 < imin(rt->tile_row.end >> 1, (by8 & ~7) + 8)) {
447
1.07k
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8 - stride],
448
1.07k
                                           ref, NULL, NULL);
449
1.07k
                }
450
1.23k
            }
451
19.4k
        }
452
41.0k
    }
453
586k
    assert(*cnt <= 8);
454
455
    // top/left (which, confusingly, is part of "secondary" references)
456
586k
    int have_dummy_newmv_match;
457
586k
    if ((n_rows | n_cols) != ~0U) {
458
359k
        add_spatial_candidate(mvstack, cnt, 4, &b_top[-1], ref, gmv,
459
359k
                              &have_dummy_newmv_match, &have_row_mvs);
460
359k
    }
461
462
    // "secondary" (non-direct neighbour) top & left edges
463
    // what is different about secondary is that everything is now in 8x8 resolution
464
1.75M
    for (int n = 2; n <= 3; n++) {
465
1.17M
        if ((unsigned) n > n_rows && (unsigned) n <= max_rows) {
466
392k
            n_rows += scan_row(mvstack, cnt, ref, gmv,
467
392k
                               &rt->r[(((by4 & 31) - 2 * n + 1) | 1) + 5][bx4 | 1],
468
392k
                               bw4, w4, 1 + max_rows - n, bw4 >= 16 ? 4 : 2,
469
392k
                               &have_dummy_newmv_match, &have_row_mvs);
470
392k
        }
471
472
1.17M
        if ((unsigned) n > n_cols && (unsigned) n <= max_cols) {
473
503k
            n_cols += scan_col(mvstack, cnt, ref, gmv, &rt->r[((by4 & 31) | 1) + 5],
474
503k
                               bh4, h4, (bx4 - n * 2 + 1) | 1,
475
503k
                               1 + max_cols - n, bh4 >= 16 ? 4 : 2,
476
503k
                               &have_dummy_newmv_match, &have_col_mvs);
477
503k
        }
478
1.17M
    }
479
586k
    assert(*cnt <= 8);
480
481
586k
    const int ref_match_count = have_col_mvs + have_row_mvs;
482
483
    // context build-up
484
586k
    int refmv_ctx, newmv_ctx;
485
586k
    switch (nearest_match) {
486
189k
    case 0:
487
189k
        refmv_ctx = imin(2, ref_match_count);
488
189k
        newmv_ctx = ref_match_count > 0;
489
189k
        break;
490
200k
    case 1:
491
200k
        refmv_ctx = imin(ref_match_count * 3, 4);
492
200k
        newmv_ctx = 3 - have_newmv;
493
200k
        break;
494
196k
    case 2:
495
196k
        refmv_ctx = 5;
496
196k
        newmv_ctx = 5 - have_newmv;
497
196k
        break;
498
586k
    }
499
500
    // sorting (nearest, then "secondary")
501
586k
    int len = nearest_cnt;
502
1.08M
    while (len) {
503
497k
        int last = 0;
504
754k
        for (int n = 1; n < len; n++) {
505
257k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
506
150k
#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
507
107k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
508
107k
                last = n;
509
107k
            }
510
257k
        }
511
497k
        len = last;
512
497k
    }
513
586k
    len = *cnt;
514
876k
    while (len > nearest_cnt) {
515
289k
        int last = nearest_cnt;
516
448k
        for (int n = nearest_cnt + 1; n < len; n++) {
517
158k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
518
43.0k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
519
43.0k
#undef EXCHANGE
520
43.0k
                last = n;
521
43.0k
            }
522
158k
        }
523
289k
        len = last;
524
289k
    }
525
526
586k
    if (ref.ref[1] > 0) {
527
96.8k
        if (*cnt < 2) {
528
78.1k
            const int sign0 = rf->sign_bias[ref.ref[0] - 1];
529
78.1k
            const int sign1 = rf->sign_bias[ref.ref[1] - 1];
530
78.1k
            const int sz4 = imin(w4, h4);
531
78.1k
            refmvs_candidate *const same = &mvstack[*cnt];
532
78.1k
            int same_count[4] = { 0 };
533
534
            // non-self references in top
535
92.4k
            if (n_rows != ~0U) for (int x = 0; x < sz4;) {
536
48.1k
                const refmvs_block *const cand_b = &b_top[x];
537
48.1k
                add_compound_extended_candidate(same, same_count, cand_b,
538
48.1k
                                                sign0, sign1, ref, rf->sign_bias);
539
48.1k
                x += dav1d_block_dimensions[cand_b->bs][0];
540
48.1k
            }
541
542
            // non-self references in left
543
92.2k
            if (n_cols != ~0U) for (int y = 0; y < sz4;) {
544
49.3k
                const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
545
49.3k
                add_compound_extended_candidate(same, same_count, cand_b,
546
49.3k
                                                sign0, sign1, ref, rf->sign_bias);
547
49.3k
                y += dav1d_block_dimensions[cand_b->bs][1];
548
49.3k
            }
549
550
78.1k
            refmvs_candidate *const diff = &same[2];
551
78.1k
            const int *const diff_count = &same_count[2];
552
553
            // merge together
554
234k
            for (int n = 0; n < 2; n++) {
555
156k
                int m = same_count[n];
556
557
156k
                if (m >= 2) continue;
558
559
129k
                const int l = diff_count[n];
560
129k
                if (l) {
561
71.5k
                    same[m].mv.mv[n] = diff[0].mv.mv[n];
562
71.5k
                    if (++m == 2) continue;
563
24.5k
                    if (l == 2) {
564
20.1k
                        same[1].mv.mv[n] = diff[1].mv.mv[n];
565
20.1k
                        continue;
566
20.1k
                    }
567
24.5k
                }
568
116k
                do {
569
116k
                    same[m].mv.mv[n] = tgmv[n];
570
116k
                } while (++m < 2);
571
62.2k
            }
572
573
            // if the first extended was the same as the non-extended one,
574
            // then replace it with the second extended one
575
78.1k
            int n = *cnt;
576
78.1k
            if (n == 1 && mvstack[0].mv.n == same[0].mv.n)
577
19.9k
                mvstack[1].mv = mvstack[2].mv;
578
126k
            do {
579
126k
                mvstack[n].weight = 2;
580
126k
            } while (++n < 2);
581
78.1k
            *cnt = 2;
582
78.1k
        }
583
584
        // clamping
585
96.8k
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
586
96.8k
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
587
96.8k
        const int top = -(by4 + bh4 + 4) * 4 * 8;
588
96.8k
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
589
590
96.8k
        const int n_refmvs = *cnt;
591
96.8k
        int n = 0;
592
203k
        do {
593
203k
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
594
203k
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
595
203k
            mvstack[n].mv.mv[1].x = iclip(mvstack[n].mv.mv[1].x, left, right);
596
203k
            mvstack[n].mv.mv[1].y = iclip(mvstack[n].mv.mv[1].y, top, bottom);
597
203k
        } while (++n < n_refmvs);
598
599
96.8k
        switch (refmv_ctx >> 1) {
600
54.9k
        case 0:
601
54.9k
            *ctx = imin(newmv_ctx, 1);
602
54.9k
            break;
603
25.3k
        case 1:
604
25.3k
            *ctx = 1 + imin(newmv_ctx, 3);
605
25.3k
            break;
606
16.4k
        case 2:
607
16.4k
            *ctx = iclip(3 + newmv_ctx, 4, 7);
608
16.4k
            break;
609
96.8k
        }
610
611
96.8k
        return;
612
489k
    } else if (*cnt < 2 && ref.ref[0] > 0) {
613
180k
        const int sign = rf->sign_bias[ref.ref[0] - 1];
614
180k
        const int sz4 = imin(w4, h4);
615
616
        // non-self references in top
617
185k
        if (n_rows != ~0U) for (int x = 0; x < sz4 && *cnt < 2;) {
618
93.9k
            const refmvs_block *const cand_b = &b_top[x];
619
93.9k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
620
93.9k
            x += dav1d_block_dimensions[cand_b->bs][0];
621
93.9k
        }
622
623
        // non-self references in left
624
180k
        if (n_cols != ~0U) for (int y = 0; y < sz4 && *cnt < 2;) {
625
83.0k
            const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
626
83.0k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
627
83.0k
            y += dav1d_block_dimensions[cand_b->bs][1];
628
83.0k
        }
629
180k
    }
630
489k
    assert(*cnt <= 8);
631
632
    // clamping
633
489k
    int n_refmvs = *cnt;
634
489k
    if (n_refmvs) {
635
395k
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
636
395k
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
637
395k
        const int top = -(by4 + bh4 + 4) * 4 * 8;
638
395k
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
639
640
395k
        int n = 0;
641
994k
        do {
642
994k
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
643
994k
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
644
994k
        } while (++n < n_refmvs);
645
395k
    }
646
647
793k
    for (int n = *cnt; n < 2; n++)
648
304k
        mvstack[n].mv.mv[0] = tgmv[0];
649
650
489k
    *ctx = (refmv_ctx << 4) | (globalmv_ctx << 3) | newmv_ctx;
651
489k
}
652
653
void dav1d_refmvs_tile_sbrow_init(refmvs_tile *const rt, const refmvs_frame *const rf,
654
                                  const int tile_col_start4, const int tile_col_end4,
655
                                  const int tile_row_start4, const int tile_row_end4,
656
                                  const int sby, int tile_row_idx, const int pass)
657
233k
{
658
233k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
659
233k
    rt->rp_proj = &rf->rp_proj[16 * rf->rp_stride * tile_row_idx];
660
233k
    const ptrdiff_t r_stride = rf->rp_stride * 2;
661
233k
    const ptrdiff_t pass_off = (rf->n_frame_threads > 1 && pass == 2) ?
662
121k
        35 * 2 * rf->n_blocks : 0;
663
233k
    refmvs_block *r = &rf->r[35 * r_stride * tile_row_idx + pass_off];
664
233k
    const int sbsz = rf->sbsz;
665
233k
    const int off = (sbsz * sby) & 16;
666
5.14M
    for (int i = 0; i < sbsz; i++, r += r_stride)
667
4.90M
        rt->r[off + 5 + i] = r;
668
233k
    rt->r[off + 0] = r;
669
233k
    r += r_stride;
670
233k
    rt->r[off + 1] = NULL;
671
233k
    rt->r[off + 2] = r;
672
233k
    r += r_stride;
673
233k
    rt->r[off + 3] = NULL;
674
233k
    rt->r[off + 4] = r;
675
233k
    if (sby & 1) {
676
74.5k
#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
677
24.8k
        EXCHANGE(rt->r[off + 0], rt->r[off + sbsz + 0]);
678
24.8k
        EXCHANGE(rt->r[off + 2], rt->r[off + sbsz + 2]);
679
24.8k
        EXCHANGE(rt->r[off + 4], rt->r[off + sbsz + 4]);
680
24.8k
#undef EXCHANGE
681
24.8k
    }
682
683
233k
    rt->rf = rf;
684
233k
    rt->tile_row.start = tile_row_start4;
685
233k
    rt->tile_row.end = imin(tile_row_end4, rf->ih4);
686
233k
    rt->tile_col.start = tile_col_start4;
687
233k
    rt->tile_col.end = imin(tile_col_end4, rf->iw4);
688
233k
}
689
690
static void load_tmvs_c(const refmvs_frame *const rf, int tile_row_idx,
691
                        const int col_start8, const int col_end8,
692
                        const int row_start8, int row_end8)
693
21.0k
{
694
21.0k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
695
21.0k
    assert(row_start8 >= 0);
696
21.0k
    assert((unsigned) (row_end8 - row_start8) <= 16U);
697
21.0k
    row_end8 = imin(row_end8, rf->ih8);
698
21.0k
    const int col_start8i = imax(col_start8 - 8, 0);
699
21.0k
    const int col_end8i = imin(col_end8 + 8, rf->iw8);
700
701
21.0k
    const ptrdiff_t stride = rf->rp_stride;
702
21.0k
    refmvs_temporal_block *rp_proj =
703
21.0k
        &rf->rp_proj[16 * stride * tile_row_idx + (row_start8 & 15) * stride];
704
164k
    for (int y = row_start8; y < row_end8; y++) {
705
319k
        for (int x = col_start8; x < col_end8; x++)
706
176k
            rp_proj[x].mv.n = INVALID_MV;
707
143k
        rp_proj += stride;
708
143k
    }
709
710
21.0k
    rp_proj = &rf->rp_proj[16 * stride * tile_row_idx];
711
48.1k
    for (int n = 0; n < rf->n_mfmvs; n++) {
712
27.0k
        const int ref2cur = rf->mfmv_ref2cur[n];
713
27.0k
        if (ref2cur == INVALID_REF2CUR) continue;
714
715
25.0k
        const int ref = rf->mfmv_ref[n];
716
25.0k
        const int ref_sign = ref - 4;
717
25.0k
        const refmvs_temporal_block *r = &rf->rp_ref[ref][row_start8 * stride];
718
188k
        for (int y = row_start8; y < row_end8; y++) {
719
162k
            const int y_sb_align = y & ~7;
720
162k
            const int y_proj_start = imax(y_sb_align, row_start8);
721
162k
            const int y_proj_end = imin(y_sb_align + 8, row_end8);
722
342k
            for (int x = col_start8i; x < col_end8i; x++) {
723
179k
                const refmvs_temporal_block *rb = &r[x];
724
179k
                const int b_ref = rb->ref;
725
179k
                if (!b_ref) continue;
726
57.7k
                const int ref2ref = rf->mfmv_ref2ref[n][b_ref - 1];
727
57.7k
                if (!ref2ref) continue;
728
51.2k
                const mv b_mv = rb->mv;
729
51.2k
                const mv offset = mv_projection(b_mv, ref2cur, ref2ref);
730
51.2k
                int pos_x = x + apply_sign(abs(offset.x) >> 6,
731
51.2k
                                           offset.x ^ ref_sign);
732
51.2k
                const int pos_y = y + apply_sign(abs(offset.y) >> 6,
733
51.2k
                                                 offset.y ^ ref_sign);
734
51.2k
                if (pos_y >= y_proj_start && pos_y < y_proj_end) {
735
46.0k
                    const ptrdiff_t pos = (pos_y & 15) * stride;
736
64.1k
                    for (;;) {
737
64.1k
                        const int x_sb_align = x & ~7;
738
64.1k
                        if (pos_x >= imax(x_sb_align - 8, col_start8) &&
739
63.8k
                            pos_x < imin(x_sb_align + 16, col_end8))
740
63.4k
                        {
741
63.4k
                            rp_proj[pos + pos_x].mv = rb->mv;
742
63.4k
                            rp_proj[pos + pos_x].ref = ref2ref;
743
63.4k
                        }
744
64.1k
                        if (++x >= col_end8i) break;
745
26.2k
                        rb++;
746
26.2k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
747
18.0k
                        pos_x++;
748
18.0k
                    }
749
46.0k
                } else {
750
5.60k
                    for (;;) {
751
5.60k
                        if (++x >= col_end8i) break;
752
2.94k
                        rb++;
753
2.94k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
754
2.94k
                    }
755
5.10k
                }
756
51.2k
                x--;
757
51.2k
            }
758
162k
            r += stride;
759
162k
        }
760
25.0k
    }
761
21.0k
}
762
763
static void save_tmvs_c(refmvs_temporal_block *rp, const ptrdiff_t stride,
764
                        refmvs_block *const *const rr,
765
                        const uint8_t *const ref_sign,
766
                        const int col_end8, const int row_end8,
767
                        const int col_start8, const int row_start8)
768
43.7k
{
769
251k
    for (int y = row_start8; y < row_end8; y++) {
770
207k
        const refmvs_block *const b = rr[(y & 15) * 2];
771
772
424k
        for (int x = col_start8; x < col_end8;) {
773
217k
            const refmvs_block *const cand_b = &b[x * 2 + 1];
774
217k
            const int bw8 = (dav1d_block_dimensions[cand_b->bs][0] + 1) >> 1;
775
776
217k
            if (cand_b->ref.ref[1] > 0 && ref_sign[cand_b->ref.ref[1] - 1] &&
777
12.3k
                (abs(cand_b->mv.mv[1].y) | abs(cand_b->mv.mv[1].x)) < 4096)
778
12.0k
            {
779
12.0k
                const refmvs_temporal_block tmv = {
780
12.0k
                    .mv = cand_b->mv.mv[1],
781
12.0k
                    .ref = cand_b->ref.ref[1],
782
12.0k
                };
783
56.9k
                for (int n = 0; n < bw8; n++, x++)
784
44.8k
                    rp[x] = tmv;
785
205k
            } else if (cand_b->ref.ref[0] > 0 && ref_sign[cand_b->ref.ref[0] - 1] &&
786
30.6k
                       (abs(cand_b->mv.mv[0].y) | abs(cand_b->mv.mv[0].x)) < 4096)
787
30.3k
            {
788
30.3k
                const refmvs_temporal_block tmv = {
789
30.3k
                    .mv = cand_b->mv.mv[0],
790
30.3k
                    .ref = cand_b->ref.ref[0],
791
30.3k
                };
792
203k
                for (int n = 0; n < bw8; n++, x++)
793
173k
                    rp[x] = tmv;
794
174k
            } else {
795
174k
                const refmvs_temporal_block tmv = { .mv = { .n = 0 }, .ref = 0 };
796
1.08M
                for (int n = 0; n < bw8; n++, x++)
797
910k
                    rp[x] = tmv;
798
174k
            }
799
217k
        }
800
207k
        rp += stride;
801
207k
    }
802
43.7k
}
803
804
int dav1d_refmvs_init_frame(refmvs_frame *const rf,
805
                            const Dav1dSequenceHeader *const seq_hdr,
806
                            const Dav1dFrameHeader *const frm_hdr,
807
                            const uint8_t ref_poc[7],
808
                            refmvs_temporal_block *const rp,
809
                            const uint8_t ref_ref_poc[7][7],
810
                            /*const*/ refmvs_temporal_block *const rp_ref[7],
811
                            const int n_tile_threads, const int n_frame_threads)
812
102k
{
813
102k
    const int rp_stride = ((frm_hdr->width[0] + 127) & ~127) >> 3;
814
102k
    const int n_tile_rows = n_tile_threads > 1 ? frm_hdr->tiling.rows : 1;
815
102k
    const int n_blocks = rp_stride * n_tile_rows;
816
817
102k
    rf->sbsz = 16 << seq_hdr->sb128;
818
102k
    rf->frm_hdr = frm_hdr;
819
102k
    rf->iw8 = (frm_hdr->width[0] + 7) >> 3;
820
102k
    rf->ih8 = (frm_hdr->height + 7) >> 3;
821
102k
    rf->iw4 = rf->iw8 << 1;
822
102k
    rf->ih4 = rf->ih8 << 1;
823
102k
    rf->rp = rp;
824
102k
    rf->rp_stride = rp_stride;
825
102k
    rf->n_tile_threads = n_tile_threads;
826
102k
    rf->n_frame_threads = n_frame_threads;
827
828
102k
    if (n_blocks != rf->n_blocks) {
829
28.1k
        const size_t r_sz = sizeof(*rf->r) * 35 * 2 * n_blocks * (1 + (n_frame_threads > 1));
830
28.1k
        const size_t rp_proj_sz = sizeof(*rf->rp_proj) * 16 * n_blocks;
831
        /* Note that sizeof(*rf->r) == 12, but it's accessed using 16-byte unaligned
832
         * loads in save_tmvs() asm which can overread 4 bytes into rp_proj. */
833
28.1k
        dav1d_free_aligned(rf->r);
834
28.1k
        rf->r = dav1d_alloc_aligned(ALLOC_REFMVS, r_sz + rp_proj_sz, 64);
835
28.1k
        if (!rf->r) {
836
0
            rf->n_blocks = 0;
837
0
            return DAV1D_ERR(ENOMEM);
838
0
        }
839
840
28.1k
        rf->rp_proj = (refmvs_temporal_block*)((uintptr_t)rf->r + r_sz);
841
28.1k
        rf->n_blocks = n_blocks;
842
28.1k
    }
843
844
102k
    const int poc = frm_hdr->frame_offset;
845
817k
    for (int i = 0; i < 7; i++) {
846
715k
        const int poc_diff = get_poc_diff(seq_hdr->order_hint_n_bits,
847
715k
                                          ref_poc[i], poc);
848
715k
        rf->sign_bias[i] = poc_diff > 0;
849
715k
        rf->mfmv_sign[i] = poc_diff < 0;
850
715k
        rf->pocdiff[i] = iclip(get_poc_diff(seq_hdr->order_hint_n_bits,
851
715k
                                            poc, ref_poc[i]), -31, 31);
852
715k
    }
853
854
    // temporal MV setup
855
102k
    rf->n_mfmvs = 0;
856
102k
    rf->rp_ref = rp_ref;
857
102k
    if (frm_hdr->use_ref_frame_mvs && seq_hdr->order_hint_n_bits) {
858
20.9k
        int total = 2;
859
20.9k
        if (rp_ref[0] && ref_ref_poc[0][6] != ref_poc[3] /* alt-of-last != gold */) {
860
6.15k
            rf->mfmv_ref[rf->n_mfmvs++] = 0; // last
861
6.15k
            total = 3;
862
6.15k
        }
863
20.9k
        if (rp_ref[4] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[4],
864
16.1k
                                      frm_hdr->frame_offset) > 0)
865
1.27k
        {
866
1.27k
            rf->mfmv_ref[rf->n_mfmvs++] = 4; // bwd
867
1.27k
        }
868
20.9k
        if (rp_ref[5] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[5],
869
17.7k
                                      frm_hdr->frame_offset) > 0)
870
362
        {
871
362
            rf->mfmv_ref[rf->n_mfmvs++] = 5; // altref2
872
362
        }
873
20.9k
        if (rf->n_mfmvs < total && rp_ref[6] &&
874
8.98k
            get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[6],
875
8.98k
                         frm_hdr->frame_offset) > 0)
876
2.49k
        {
877
2.49k
            rf->mfmv_ref[rf->n_mfmvs++] = 6; // altref
878
2.49k
        }
879
20.9k
        if (rf->n_mfmvs < total && rp_ref[1])
880
17.0k
            rf->mfmv_ref[rf->n_mfmvs++] = 1; // last2
881
882
48.2k
        for (int n = 0; n < rf->n_mfmvs; n++) {
883
27.3k
            const int rpoc = ref_poc[rf->mfmv_ref[n]];
884
27.3k
            const int diff1 = get_poc_diff(seq_hdr->order_hint_n_bits,
885
27.3k
                                           rpoc, frm_hdr->frame_offset);
886
27.3k
            if (abs(diff1) > 31) {
887
1.75k
                rf->mfmv_ref2cur[n] = INVALID_REF2CUR;
888
25.5k
            } else {
889
25.5k
                rf->mfmv_ref2cur[n] = rf->mfmv_ref[n] < 4 ? -diff1 : diff1;
890
204k
                for (int m = 0; m < 7; m++) {
891
178k
                    const int rrpoc = ref_ref_poc[rf->mfmv_ref[n]][m];
892
178k
                    const int diff2 = get_poc_diff(seq_hdr->order_hint_n_bits,
893
178k
                                                   rpoc, rrpoc);
894
                    // unsigned comparison also catches the < 0 case
895
178k
                    rf->mfmv_ref2ref[n][m] = (unsigned) diff2 > 31U ? 0 : diff2;
896
178k
                }
897
25.5k
            }
898
27.3k
        }
899
20.9k
    }
900
102k
    rf->use_ref_frame_mvs = rf->n_mfmvs > 0;
901
902
102k
    return 0;
903
102k
}
904
905
static void splat_mv_c(refmvs_block **rr, const refmvs_block *const rmv,
906
                       const int bx4, const int bw4, int bh4)
907
1.32M
{
908
5.72M
    do {
909
5.72M
        refmvs_block *const r = *rr++ + bx4;
910
57.6M
        for (int x = 0; x < bw4; x++)
911
51.9M
            r[x] = *rmv;
912
5.72M
    } while (--bh4);
913
1.32M
}
914
915
#if HAVE_ASM
916
#if ARCH_AARCH64 || ARCH_ARM
917
#include "src/arm/refmvs.h"
918
#elif ARCH_LOONGARCH64
919
#include "src/loongarch/refmvs.h"
920
#elif ARCH_X86
921
#include "src/x86/refmvs.h"
922
#endif
923
#endif
924
925
COLD void dav1d_refmvs_dsp_init(Dav1dRefmvsDSPContext *const c)
926
24.6k
{
927
24.6k
    c->load_tmvs = load_tmvs_c;
928
24.6k
    c->save_tmvs = save_tmvs_c;
929
24.6k
    c->splat_mv = splat_mv_c;
930
931
#if HAVE_ASM
932
#if ARCH_AARCH64 || ARCH_ARM
933
    refmvs_dsp_init_arm(c);
934
#elif ARCH_LOONGARCH64
935
    refmvs_dsp_init_loongarch(c);
936
#elif ARCH_X86
937
    refmvs_dsp_init_x86(c);
938
#endif
939
#endif
940
24.6k
}