Coverage Report

Created: 2026-07-30 06:22

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/dav1d/src/refmvs.c
Line
Count
Source
1
/*
2
 * Copyright © 2020, VideoLAN and dav1d authors
3
 * Copyright © 2020, Two Orioles, LLC
4
 * All rights reserved.
5
 *
6
 * Redistribution and use in source and binary forms, with or without
7
 * modification, are permitted provided that the following conditions are met:
8
 *
9
 * 1. Redistributions of source code must retain the above copyright notice, this
10
 *    list of conditions and the following disclaimer.
11
 *
12
 * 2. Redistributions in binary form must reproduce the above copyright notice,
13
 *    this list of conditions and the following disclaimer in the documentation
14
 *    and/or other materials provided with the distribution.
15
 *
16
 * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
17
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
18
 * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
19
 * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
20
 * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
21
 * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
22
 * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
23
 * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
24
 * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
25
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
26
 */
27
28
#include "config.h"
29
30
#include <limits.h>
31
#include <stdlib.h>
32
33
#include "dav1d/common.h"
34
35
#include "common/intops.h"
36
37
#include "src/env.h"
38
#include "src/mem.h"
39
#include "src/refmvs.h"
40
41
static void add_spatial_candidate(refmvs_candidate *const mvstack, int *const cnt,
42
                                  const int weight, const refmvs_block *const b,
43
                                  const union refmvs_refpair ref, const mv gmv[2],
44
                                  int *const have_newmv_match,
45
                                  int *const have_refmv_match)
46
2.46M
{
47
2.46M
    if (b->mv.mv[0].n == INVALID_MV) return; // intra block, no intrabc
48
49
2.27M
    if (ref.ref[1] == -1) {
50
2.71M
        for (int n = 0; n < 2; n++) {
51
2.39M
            if (b->ref.ref[n] == ref.ref[0]) {
52
1.73M
                const mv cand_mv = ((b->mf & 1) && gmv[0].n != INVALID_MV) ?
53
1.68M
                                   gmv[0] : b->mv.mv[n];
54
55
1.73M
                *have_refmv_match = 1;
56
1.73M
                *have_newmv_match |= b->mf >> 1;
57
58
1.73M
                const int last = *cnt;
59
3.03M
                for (int m = 0; m < last; m++)
60
1.88M
                    if (mvstack[m].mv.mv[0].n == cand_mv.n) {
61
582k
                        mvstack[m].weight += weight;
62
582k
                        return;
63
582k
                    }
64
65
1.15M
                if (last < 8) {
66
1.15M
                    mvstack[last].mv.mv[0] = cand_mv;
67
1.15M
                    mvstack[last].weight = weight;
68
1.15M
                    *cnt = last + 1;
69
1.15M
                }
70
1.15M
                return;
71
1.73M
            }
72
2.39M
        }
73
2.04M
    } else if (b->ref.pair == ref.pair) {
74
78.6k
        const refmvs_mvpair cand_mv = { .mv = {
75
78.6k
            [0] = ((b->mf & 1) && gmv[0].n != INVALID_MV) ? gmv[0] : b->mv.mv[0],
76
78.6k
            [1] = ((b->mf & 1) && gmv[1].n != INVALID_MV) ? gmv[1] : b->mv.mv[1],
77
78.6k
        }};
78
79
78.6k
        *have_refmv_match = 1;
80
78.6k
        *have_newmv_match |= b->mf >> 1;
81
82
78.6k
        const int last = *cnt;
83
118k
        for (int n = 0; n < last; n++)
84
66.9k
            if (mvstack[n].mv.n == cand_mv.n) {
85
27.4k
                mvstack[n].weight += weight;
86
27.4k
                return;
87
27.4k
            }
88
89
51.1k
        if (last < 8) {
90
51.1k
            mvstack[last].mv = cand_mv;
91
51.1k
            mvstack[last].weight = weight;
92
51.1k
            *cnt = last + 1;
93
51.1k
        }
94
51.1k
    }
95
2.27M
}
96
97
static int scan_row(refmvs_candidate *const mvstack, int *const cnt,
98
                    const union refmvs_refpair ref, const mv gmv[2],
99
                    const refmvs_block *b, const int bw4, const int w4,
100
                    const int max_rows, const int step,
101
                    int *const have_newmv_match, int *const have_refmv_match)
102
752k
{
103
752k
    const refmvs_block *cand_b = b;
104
752k
    const enum BlockSize first_cand_bs = cand_b->bs;
105
752k
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
106
752k
    int cand_bw4 = first_cand_b_dim[0];
107
752k
    int len = imax(step, imin(bw4, cand_bw4));
108
109
752k
    if (bw4 <= cand_bw4) {
110
        // FIXME weight can be higher for odd blocks (bx4 & 1), but then the
111
        // position of the first block has to be odd already, i.e. not just
112
        // for row_offset=-3/-5
113
        // FIXME why can this not be cand_bw4?
114
667k
        const int weight = bw4 == 1 ? 2 :
115
667k
                           imax(2, imin(2 * max_rows, first_cand_b_dim[1]));
116
667k
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
117
667k
                              have_newmv_match, have_refmv_match);
118
667k
        return weight >> 1;
119
667k
    }
120
121
151k
    for (int x = 0;;) {
122
        // FIXME if we overhang above, we could fill a bitmask so we don't have
123
        // to repeat the add_spatial_candidate() for the next row, but just increase
124
        // the weight here
125
151k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
126
151k
                              have_newmv_match, have_refmv_match);
127
151k
        x += len;
128
151k
        if (x >= w4) return 1;
129
66.3k
        cand_b = &b[x];
130
66.3k
        cand_bw4 = dav1d_block_dimensions[cand_b->bs][0];
131
66.3k
        assert(cand_bw4 < bw4);
132
66.3k
        len = imax(step, cand_bw4);
133
66.3k
    }
134
85.2k
}
135
136
static int scan_col(refmvs_candidate *const mvstack, int *const cnt,
137
                    const union refmvs_refpair ref, const mv gmv[2],
138
                    /*const*/ refmvs_block *const *b, const int bh4, const int h4,
139
                    const int bx4, const int max_cols, const int step,
140
                    int *const have_newmv_match, int *const have_refmv_match)
141
1.05M
{
142
1.05M
    const refmvs_block *cand_b = &b[0][bx4];
143
1.05M
    const enum BlockSize first_cand_bs = cand_b->bs;
144
1.05M
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
145
1.05M
    int cand_bh4 = first_cand_b_dim[1];
146
1.05M
    int len = imax(step, imin(bh4, cand_bh4));
147
148
1.05M
    if (bh4 <= cand_bh4) {
149
        // FIXME weight can be higher for odd blocks (by4 & 1), but then the
150
        // position of the first block has to be odd already, i.e. not just
151
        // for col_offset=-3/-5
152
        // FIXME why can this not be cand_bh4?
153
958k
        const int weight = bh4 == 1 ? 2 :
154
958k
                           imax(2, imin(2 * max_cols, first_cand_b_dim[0]));
155
958k
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
156
958k
                            have_newmv_match, have_refmv_match);
157
958k
        return weight >> 1;
158
958k
    }
159
160
184k
    for (int y = 0;;) {
161
        // FIXME if we overhang above, we could fill a bitmask so we don't have
162
        // to repeat the add_spatial_candidate() for the next row, but just increase
163
        // the weight here
164
184k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
165
184k
                              have_newmv_match, have_refmv_match);
166
184k
        y += len;
167
184k
        if (y >= h4) return 1;
168
89.4k
        cand_b = &b[y][bx4];
169
89.4k
        cand_bh4 = dav1d_block_dimensions[cand_b->bs][1];
170
89.4k
        assert(cand_bh4 < bh4);
171
89.4k
        len = imax(step, cand_bh4);
172
89.4k
    }
173
96.4k
}
174
175
76.0k
static inline union mv mv_projection(const union mv mv, const int num, const int den) {
176
76.0k
    static const uint16_t div_mult[32] = {
177
76.0k
           0, 16384, 8192, 5461, 4096, 3276, 2730, 2340,
178
76.0k
        2048,  1820, 1638, 1489, 1365, 1260, 1170, 1092,
179
76.0k
        1024,   963,  910,  862,  819,  780,  744,  712,
180
76.0k
         682,   655,  630,  606,  585,  564,  546,  528
181
76.0k
    };
182
76.0k
    assert(den > 0 && den < 32);
183
76.0k
    assert(num > -32 && num < 32);
184
76.0k
    const int frac = num * div_mult[den];
185
76.0k
    const int y = mv.y * frac, x = mv.x * frac;
186
    // Round and clip according to AV1 spec section 7.9.3
187
76.0k
    return (union mv) { // 0x3fff == (1 << 14) - 1
188
76.0k
        .y = iclip((y + 8192 + (y >> 31)) >> 14, -0x3fff, 0x3fff),
189
76.0k
        .x = iclip((x + 8192 + (x >> 31)) >> 14, -0x3fff, 0x3fff)
190
76.0k
    };
191
76.0k
}
192
193
static void add_temporal_candidate(const refmvs_frame *const rf,
194
                                   refmvs_candidate *const mvstack, int *const cnt,
195
                                   const refmvs_temporal_block *const rb,
196
                                   const union refmvs_refpair ref, int *const globalmv_ctx,
197
                                   const union mv gmv[])
198
65.2k
{
199
65.2k
    if (rb->mv.n == INVALID_MV) return;
200
201
45.1k
    union mv mv = mv_projection(rb->mv, rf->pocdiff[ref.ref[0] - 1], rb->ref);
202
45.1k
    fix_mv_precision(rf->frm_hdr, &mv);
203
204
45.1k
    const int last = *cnt;
205
45.1k
    if (ref.ref[1] == -1) {
206
30.5k
        if (globalmv_ctx)
207
6.73k
            *globalmv_ctx = (abs(mv.x - gmv[0].x) | abs(mv.y - gmv[0].y)) >= 16;
208
209
49.6k
        for (int n = 0; n < last; n++)
210
43.8k
            if (mvstack[n].mv.mv[0].n == mv.n) {
211
24.7k
                mvstack[n].weight += 2;
212
24.7k
                return;
213
24.7k
            }
214
5.80k
        if (last < 8) {
215
5.75k
            mvstack[last].mv.mv[0] = mv;
216
5.75k
            mvstack[last].weight = 2;
217
5.75k
            *cnt = last + 1;
218
5.75k
        }
219
14.6k
    } else {
220
14.6k
        refmvs_mvpair mvp = { .mv = {
221
14.6k
            [0] = mv,
222
14.6k
            [1] = mv_projection(rb->mv, rf->pocdiff[ref.ref[1] - 1], rb->ref),
223
14.6k
        }};
224
14.6k
        fix_mv_precision(rf->frm_hdr, &mvp.mv[1]);
225
226
23.1k
        for (int n = 0; n < last; n++)
227
20.2k
            if (mvstack[n].mv.n == mvp.n) {
228
11.7k
                mvstack[n].weight += 2;
229
11.7k
                return;
230
11.7k
            }
231
2.90k
        if (last < 8) {
232
2.86k
            mvstack[last].mv = mvp;
233
2.86k
            mvstack[last].weight = 2;
234
2.86k
            *cnt = last + 1;
235
2.86k
        }
236
2.90k
    }
237
45.1k
}
238
239
static void add_compound_extended_candidate(refmvs_candidate *const same,
240
                                            int *const same_count,
241
                                            const refmvs_block *const cand_b,
242
                                            const int sign0, const int sign1,
243
                                            const union refmvs_refpair ref,
244
                                            const uint8_t *const sign_bias)
245
63.6k
{
246
63.6k
    refmvs_candidate *const diff = &same[2];
247
63.6k
    int *const diff_count = &same_count[2];
248
249
159k
    for (int n = 0; n < 2; n++) {
250
124k
        const int cand_ref = cand_b->ref.ref[n];
251
252
124k
        if (cand_ref <= 0) break;
253
254
96.2k
        mv cand_mv = cand_b->mv.mv[n];
255
96.2k
        if (cand_ref == ref.ref[0]) {
256
35.1k
            if (same_count[0] < 2)
257
34.0k
                same[same_count[0]++].mv.mv[0] = cand_mv;
258
35.1k
            if (diff_count[1] < 2) {
259
30.8k
                if (sign1 ^ sign_bias[cand_ref - 1]) {
260
1.69k
                    cand_mv.y = -cand_mv.y;
261
1.69k
                    cand_mv.x = -cand_mv.x;
262
1.69k
                }
263
30.8k
                diff[diff_count[1]++].mv.mv[1] = cand_mv;
264
30.8k
            }
265
61.1k
        } else if (cand_ref == ref.ref[1]) {
266
34.2k
            if (same_count[1] < 2)
267
33.5k
                same[same_count[1]++].mv.mv[1] = cand_mv;
268
34.2k
            if (diff_count[0] < 2) {
269
29.2k
                if (sign0 ^ sign_bias[cand_ref - 1]) {
270
1.75k
                    cand_mv.y = -cand_mv.y;
271
1.75k
                    cand_mv.x = -cand_mv.x;
272
1.75k
                }
273
29.2k
                diff[diff_count[0]++].mv.mv[0] = cand_mv;
274
29.2k
            }
275
34.2k
        } else {
276
26.8k
            mv i_cand_mv = (union mv) {
277
26.8k
                .x = -cand_mv.x,
278
26.8k
                .y = -cand_mv.y
279
26.8k
            };
280
281
26.8k
            if (diff_count[0] < 2) {
282
21.5k
                diff[diff_count[0]++].mv.mv[0] =
283
21.5k
                    sign0 ^ sign_bias[cand_ref - 1] ?
284
20.9k
                    i_cand_mv : cand_mv;
285
21.5k
            }
286
287
26.8k
            if (diff_count[1] < 2) {
288
20.4k
                diff[diff_count[1]++].mv.mv[1] =
289
20.4k
                    sign1 ^ sign_bias[cand_ref - 1] ?
290
19.8k
                    i_cand_mv : cand_mv;
291
20.4k
            }
292
26.8k
        }
293
96.2k
    }
294
63.6k
}
295
296
static void add_single_extended_candidate(refmvs_candidate mvstack[8], int *const cnt,
297
                                          const refmvs_block *const cand_b,
298
                                          const int sign, const uint8_t *const sign_bias)
299
302k
{
300
607k
    for (int n = 0; n < 2; n++) {
301
594k
        const int cand_ref = cand_b->ref.ref[n];
302
303
594k
        if (cand_ref <= 0) break;
304
        // we need to continue even if cand_ref == ref.ref[0], since
305
        // the candidate could have been added as a globalmv variant,
306
        // which changes the value
307
        // FIXME if scan_{row,col}() returned a mask for the nearest
308
        // edge, we could skip the appropriate ones here
309
310
304k
        mv cand_mv = cand_b->mv.mv[n];
311
304k
        if (sign ^ sign_bias[cand_ref - 1]) {
312
6.74k
            cand_mv.y = -cand_mv.y;
313
6.74k
            cand_mv.x = -cand_mv.x;
314
6.74k
        }
315
316
304k
        int m;
317
304k
        const int last = *cnt;
318
335k
        for (m = 0; m < last; m++)
319
274k
            if (cand_mv.n == mvstack[m].mv.mv[0].n)
320
243k
                break;
321
304k
        if (m == last) {
322
60.5k
            mvstack[m].mv.mv[0] = cand_mv;
323
60.5k
            mvstack[m].weight = 2; // "minimal"
324
60.5k
            *cnt = last + 1;
325
60.5k
        }
326
304k
    }
327
302k
}
328
329
/*
330
 * refmvs_frame allocates memory for one sbrow (32 blocks high, whole frame
331
 * wide) of 4x4-resolution refmvs_block entries for spatial MV referencing.
332
 * mvrefs_tile[] keeps a list of 35 (32 + 3 above) pointers into this memory,
333
 * and each sbrow, the bottom entries (y=27/29/31) are exchanged with the top
334
 * (-5/-3/-1) pointers by calling dav1d_refmvs_tile_sbrow_init() at the start
335
 * of each tile/sbrow.
336
 *
337
 * For temporal MV referencing, we call dav1d_refmvs_save_tmvs() at the end of
338
 * each tile/sbrow (when tile column threading is enabled), or at the start of
339
 * each interleaved sbrow (i.e. once for all tile columns together, when tile
340
 * column threading is disabled). This will copy the 4x4-resolution spatial MVs
341
 * into 8x8-resolution refmvs_temporal_block structures. Then, for subsequent
342
 * frames, at the start of each tile/sbrow (when tile column threading is
343
 * enabled) or at the start of each interleaved sbrow (when tile column
344
 * threading is disabled), we call load_tmvs(), which will project the MVs to
345
 * their respective position in the current frame.
346
 */
347
348
void dav1d_refmvs_find(const refmvs_tile *const rt,
349
                       refmvs_candidate mvstack[8], int *const cnt,
350
                       int *const ctx,
351
                       const union refmvs_refpair ref, const enum BlockSize bs,
352
                       const enum EdgeFlags edge_flags,
353
                       const int by4, const int bx4)
354
689k
{
355
689k
    const refmvs_frame *const rf = rt->rf;
356
689k
    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
357
689k
    const int bw4 = b_dim[0], w4 = imin(imin(bw4, 16), rt->tile_col.end - bx4);
358
689k
    const int bh4 = b_dim[1], h4 = imin(imin(bh4, 16), rt->tile_row.end - by4);
359
689k
    mv gmv[2], tgmv[2];
360
361
689k
    *cnt = 0;
362
689k
    assert(ref.ref[0] >=  0 && ref.ref[0] <= 8 &&
363
689k
           ref.ref[1] >= -1 && ref.ref[1] <= 8);
364
689k
    if (ref.ref[0] > 0) {
365
402k
        tgmv[0] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[0] - 1],
366
402k
                             bx4, by4, bw4, bh4, rf->frm_hdr);
367
402k
        gmv[0] = rf->frm_hdr->gmv[ref.ref[0] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
368
308k
                 tgmv[0] : (mv) { .n = INVALID_MV };
369
402k
    } else {
370
287k
        tgmv[0] = (mv) { .n = 0 };
371
287k
        gmv[0] = (mv) { .n = INVALID_MV };
372
287k
    }
373
689k
    if (ref.ref[1] > 0) {
374
70.0k
        tgmv[1] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[1] - 1],
375
70.0k
                             bx4, by4, bw4, bh4, rf->frm_hdr);
376
70.0k
        gmv[1] = rf->frm_hdr->gmv[ref.ref[1] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
377
52.4k
                 tgmv[1] : (mv) { .n = INVALID_MV };
378
70.0k
    }
379
380
    // top
381
689k
    int have_newmv = 0, have_col_mvs = 0, have_row_mvs = 0;
382
689k
    unsigned max_rows = 0, n_rows = ~0;
383
689k
    const refmvs_block *b_top;
384
689k
    if (by4 > rt->tile_row.start) {
385
429k
        max_rows = imin((by4 - rt->tile_row.start + 1) >> 1, 2 + (bh4 > 1));
386
429k
        b_top = &rt->r[(by4 & 31) - 1 + 5][bx4];
387
429k
        n_rows = scan_row(mvstack, cnt, ref, gmv, b_top,
388
429k
                          bw4, w4, max_rows, bw4 >= 16 ? 4 : 1,
389
429k
                          &have_newmv, &have_row_mvs);
390
429k
    }
391
392
    // left
393
689k
    unsigned max_cols = 0, n_cols = ~0U;
394
689k
    refmvs_block *const *b_left;
395
689k
    if (bx4 > rt->tile_col.start) {
396
513k
        max_cols = imin((bx4 - rt->tile_col.start + 1) >> 1, 2 + (bw4 > 1));
397
513k
        b_left = &rt->r[(by4 & 31) + 5];
398
513k
        n_cols = scan_col(mvstack, cnt, ref, gmv, b_left,
399
513k
                          bh4, h4, bx4 - 1, max_cols, bh4 >= 16 ? 4 : 1,
400
513k
                          &have_newmv, &have_col_mvs);
401
513k
    }
402
403
    // top/right
404
689k
    if (n_rows != ~0U && edge_flags & EDGE_I444_TOP_HAS_RIGHT &&
405
255k
        imax(bw4, bh4) <= 16 && bw4 + bx4 < rt->tile_col.end)
406
177k
    {
407
177k
        add_spatial_candidate(mvstack, cnt, 4, &b_top[bw4], ref, gmv,
408
177k
                              &have_newmv, &have_row_mvs);
409
177k
    }
410
411
689k
    const int nearest_match = have_col_mvs + have_row_mvs;
412
689k
    const int nearest_cnt = *cnt;
413
1.44M
    for (int n = 0; n < nearest_cnt; n++)
414
754k
        mvstack[n].weight += 640;
415
416
    // temporal
417
689k
    int globalmv_ctx = rf->frm_hdr->use_ref_frame_mvs;
418
689k
    if (rf->use_ref_frame_mvs) {
419
16.9k
        const ptrdiff_t stride = rf->rp_stride;
420
16.9k
        const int by8 = by4 >> 1, bx8 = bx4 >> 1;
421
16.9k
        const refmvs_temporal_block *const rbi = &rt->rp_proj[(by8 & 15) * stride + bx8];
422
16.9k
        const refmvs_temporal_block *rb = rbi;
423
16.9k
        const int step_h = bw4 >= 16 ? 2 : 1, step_v = bh4 >= 16 ? 2 : 1;
424
16.9k
        const int w8 = imin((w4 + 1) >> 1, 8), h8 = imin((h4 + 1) >> 1, 8);
425
49.6k
        for (int y = 0; y < h8; y += step_v) {
426
85.8k
            for (int x = 0; x < w8; x+= step_h) {
427
53.1k
                add_temporal_candidate(rf, mvstack, cnt, &rb[x], ref,
428
53.1k
                                       !(x | y) ? &globalmv_ctx : NULL, tgmv);
429
53.1k
            }
430
32.7k
            rb += stride * step_v;
431
32.7k
        }
432
16.9k
        if (imin(bw4, bh4) >= 2 && imax(bw4, bh4) < 16) {
433
10.9k
            const int bh8 = bh4 >> 1, bw8 = bw4 >> 1;
434
10.9k
            rb = &rbi[bh8 * stride];
435
10.9k
            const int has_bottom = by8 + bh8 < imin(rt->tile_row.end >> 1,
436
10.9k
                                                    (by8 & ~7) + 8);
437
10.9k
            if (has_bottom && bx8 - 1 >= imax(rt->tile_col.start >> 1, bx8 & ~7)) {
438
3.58k
                add_temporal_candidate(rf, mvstack, cnt, &rb[-1], ref,
439
3.58k
                                       NULL, NULL);
440
3.58k
            }
441
10.9k
            if (bx8 + bw8 < imin(rt->tile_col.end >> 1, (bx8 & ~7) + 8)) {
442
5.16k
                if (has_bottom) {
443
3.45k
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8], ref,
444
3.45k
                                           NULL, NULL);
445
3.45k
                }
446
5.16k
                if (by8 + bh8 - 1 < imin(rt->tile_row.end >> 1, (by8 & ~7) + 8)) {
447
5.04k
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8 - stride],
448
5.04k
                                           ref, NULL, NULL);
449
5.04k
                }
450
5.16k
            }
451
10.9k
        }
452
16.9k
    }
453
689k
    assert(*cnt <= 8);
454
455
    // top/left (which, confusingly, is part of "secondary" references)
456
689k
    int have_dummy_newmv_match;
457
689k
    if ((n_rows | n_cols) != ~0U) {
458
330k
        add_spatial_candidate(mvstack, cnt, 4, &b_top[-1], ref, gmv,
459
330k
                              &have_dummy_newmv_match, &have_row_mvs);
460
330k
    }
461
462
    // "secondary" (non-direct neighbour) top & left edges
463
    // what is different about secondary is that everything is now in 8x8 resolution
464
2.07M
    for (int n = 2; n <= 3; n++) {
465
1.38M
        if ((unsigned) n > n_rows && (unsigned) n <= max_rows) {
466
322k
            n_rows += scan_row(mvstack, cnt, ref, gmv,
467
322k
                               &rt->r[(((by4 & 31) - 2 * n + 1) | 1) + 5][bx4 | 1],
468
322k
                               bw4, w4, 1 + max_rows - n, bw4 >= 16 ? 4 : 2,
469
322k
                               &have_dummy_newmv_match, &have_row_mvs);
470
322k
        }
471
472
1.38M
        if ((unsigned) n > n_cols && (unsigned) n <= max_cols) {
473
544k
            n_cols += scan_col(mvstack, cnt, ref, gmv, &rt->r[((by4 & 31) | 1) + 5],
474
544k
                               bh4, h4, (bx4 - n * 2 + 1) | 1,
475
544k
                               1 + max_cols - n, bh4 >= 16 ? 4 : 2,
476
544k
                               &have_dummy_newmv_match, &have_col_mvs);
477
544k
        }
478
1.38M
    }
479
689k
    assert(*cnt <= 8);
480
481
689k
    const int ref_match_count = have_col_mvs + have_row_mvs;
482
483
    // context build-up
484
689k
    int refmv_ctx, newmv_ctx;
485
689k
    switch (nearest_match) {
486
170k
    case 0:
487
170k
        refmv_ctx = imin(2, ref_match_count);
488
170k
        newmv_ctx = ref_match_count > 0;
489
170k
        break;
490
292k
    case 1:
491
292k
        refmv_ctx = imin(ref_match_count * 3, 4);
492
292k
        newmv_ctx = 3 - have_newmv;
493
292k
        break;
494
232k
    case 2:
495
232k
        refmv_ctx = 5;
496
232k
        newmv_ctx = 5 - have_newmv;
497
232k
        break;
498
689k
    }
499
500
    // sorting (nearest, then "secondary")
501
694k
    int len = nearest_cnt;
502
1.34M
    while (len) {
503
647k
        int last = 0;
504
932k
        for (int n = 1; n < len; n++) {
505
284k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
506
138k
#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
507
128k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
508
128k
                last = n;
509
128k
            }
510
284k
        }
511
647k
        len = last;
512
647k
    }
513
694k
    len = *cnt;
514
1.02M
    while (len > nearest_cnt) {
515
334k
        int last = nearest_cnt;
516
481k
        for (int n = nearest_cnt + 1; n < len; n++) {
517
146k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
518
10.5k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
519
10.5k
#undef EXCHANGE
520
10.5k
                last = n;
521
10.5k
            }
522
146k
        }
523
334k
        len = last;
524
334k
    }
525
526
694k
    if (ref.ref[1] > 0) {
527
70.0k
        if (*cnt < 2) {
528
57.0k
            const int sign0 = rf->sign_bias[ref.ref[0] - 1];
529
57.0k
            const int sign1 = rf->sign_bias[ref.ref[1] - 1];
530
57.0k
            const int sz4 = imin(w4, h4);
531
57.0k
            refmvs_candidate *const same = &mvstack[*cnt];
532
57.0k
            int same_count[4] = { 0 };
533
534
            // non-self references in top
535
61.1k
            if (n_rows != ~0U) for (int x = 0; x < sz4;) {
536
31.7k
                const refmvs_block *const cand_b = &b_top[x];
537
31.7k
                add_compound_extended_candidate(same, same_count, cand_b,
538
31.7k
                                                sign0, sign1, ref, rf->sign_bias);
539
31.7k
                x += dav1d_block_dimensions[cand_b->bs][0];
540
31.7k
            }
541
542
            // non-self references in left
543
60.3k
            if (n_cols != ~0U) for (int y = 0; y < sz4;) {
544
31.9k
                const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
545
31.9k
                add_compound_extended_candidate(same, same_count, cand_b,
546
31.9k
                                                sign0, sign1, ref, rf->sign_bias);
547
31.9k
                y += dav1d_block_dimensions[cand_b->bs][1];
548
31.9k
            }
549
550
57.0k
            refmvs_candidate *const diff = &same[2];
551
57.0k
            const int *const diff_count = &same_count[2];
552
553
            // merge together
554
171k
            for (int n = 0; n < 2; n++) {
555
114k
                int m = same_count[n];
556
557
114k
                if (m >= 2) continue;
558
559
98.9k
                const int l = diff_count[n];
560
98.9k
                if (l) {
561
52.2k
                    same[m].mv.mv[n] = diff[0].mv.mv[n];
562
52.2k
                    if (++m == 2) continue;
563
20.3k
                    if (l == 2) {
564
12.5k
                        same[1].mv.mv[n] = diff[1].mv.mv[n];
565
12.5k
                        continue;
566
12.5k
                    }
567
20.3k
                }
568
95.8k
                do {
569
95.8k
                    same[m].mv.mv[n] = tgmv[n];
570
95.8k
                } while (++m < 2);
571
54.5k
            }
572
573
            // if the first extended was the same as the non-extended one,
574
            // then replace it with the second extended one
575
57.0k
            int n = *cnt;
576
57.0k
            if (n == 1 && mvstack[0].mv.n == same[0].mv.n)
577
14.3k
                mvstack[1].mv = mvstack[2].mv;
578
94.6k
            do {
579
94.6k
                mvstack[n].weight = 2;
580
94.6k
            } while (++n < 2);
581
57.0k
            *cnt = 2;
582
57.0k
        }
583
584
        // clamping
585
70.0k
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
586
70.0k
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
587
70.0k
        const int top = -(by4 + bh4 + 4) * 4 * 8;
588
70.0k
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
589
590
70.0k
        const int n_refmvs = *cnt;
591
70.0k
        int n = 0;
592
148k
        do {
593
148k
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
594
148k
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
595
148k
            mvstack[n].mv.mv[1].x = iclip(mvstack[n].mv.mv[1].x, left, right);
596
148k
            mvstack[n].mv.mv[1].y = iclip(mvstack[n].mv.mv[1].y, top, bottom);
597
148k
        } while (++n < n_refmvs);
598
599
70.0k
        switch (refmv_ctx >> 1) {
600
41.1k
        case 0:
601
41.1k
            *ctx = imin(newmv_ctx, 1);
602
41.1k
            break;
603
19.1k
        case 1:
604
19.1k
            *ctx = 1 + imin(newmv_ctx, 3);
605
19.1k
            break;
606
9.76k
        case 2:
607
9.76k
            *ctx = iclip(3 + newmv_ctx, 4, 7);
608
9.76k
            break;
609
70.0k
        }
610
611
70.0k
        return;
612
624k
    } else if (*cnt < 2 && ref.ref[0] > 0) {
613
256k
        const int sign = rf->sign_bias[ref.ref[0] - 1];
614
256k
        const int sz4 = imin(w4, h4);
615
616
        // non-self references in top
617
310k
        if (n_rows != ~0U) for (int x = 0; x < sz4 && *cnt < 2;) {
618
157k
            const refmvs_block *const cand_b = &b_top[x];
619
157k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
620
157k
            x += dav1d_block_dimensions[cand_b->bs][0];
621
157k
        }
622
623
        // non-self references in left
624
290k
        if (n_cols != ~0U) for (int y = 0; y < sz4 && *cnt < 2;) {
625
144k
            const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
626
144k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
627
144k
            y += dav1d_block_dimensions[cand_b->bs][1];
628
144k
        }
629
256k
    }
630
624k
    assert(*cnt <= 8);
631
632
    // clamping
633
624k
    int n_refmvs = *cnt;
634
624k
    if (n_refmvs) {
635
542k
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
636
542k
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
637
542k
        const int top = -(by4 + bh4 + 4) * 4 * 8;
638
542k
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
639
640
542k
        int n = 0;
641
1.23M
        do {
642
1.23M
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
643
1.23M
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
644
1.23M
        } while (++n < n_refmvs);
645
542k
    }
646
647
980k
    for (int n = *cnt; n < 2; n++)
648
356k
        mvstack[n].mv.mv[0] = tgmv[0];
649
650
624k
    *ctx = (refmv_ctx << 4) | (globalmv_ctx << 3) | newmv_ctx;
651
624k
}
652
653
void dav1d_refmvs_tile_sbrow_init(refmvs_tile *const rt, const refmvs_frame *const rf,
654
                                  const int tile_col_start4, const int tile_col_end4,
655
                                  const int tile_row_start4, const int tile_row_end4,
656
                                  const int sby, int tile_row_idx, const int pass)
657
254k
{
658
254k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
659
254k
    rt->rp_proj = &rf->rp_proj[16 * rf->rp_stride * tile_row_idx];
660
254k
    const ptrdiff_t r_stride = rf->rp_stride * 2;
661
254k
    const ptrdiff_t pass_off = (rf->n_frame_threads > 1 && pass == 2) ?
662
131k
        35 * 2 * rf->n_blocks : 0;
663
254k
    refmvs_block *r = &rf->r[35 * r_stride * tile_row_idx + pass_off];
664
254k
    const int sbsz = rf->sbsz;
665
254k
    const int off = (sbsz * sby) & 16;
666
5.33M
    for (int i = 0; i < sbsz; i++, r += r_stride)
667
5.08M
        rt->r[off + 5 + i] = r;
668
254k
    rt->r[off + 0] = r;
669
254k
    r += r_stride;
670
254k
    rt->r[off + 1] = NULL;
671
254k
    rt->r[off + 2] = r;
672
254k
    r += r_stride;
673
254k
    rt->r[off + 3] = NULL;
674
254k
    rt->r[off + 4] = r;
675
254k
    if (sby & 1) {
676
151k
#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
677
50.3k
        EXCHANGE(rt->r[off + 0], rt->r[off + sbsz + 0]);
678
50.3k
        EXCHANGE(rt->r[off + 2], rt->r[off + sbsz + 2]);
679
50.3k
        EXCHANGE(rt->r[off + 4], rt->r[off + sbsz + 4]);
680
50.3k
#undef EXCHANGE
681
50.3k
    }
682
683
254k
    rt->rf = rf;
684
254k
    rt->tile_row.start = tile_row_start4;
685
254k
    rt->tile_row.end = imin(tile_row_end4, rf->ih4);
686
254k
    rt->tile_col.start = tile_col_start4;
687
254k
    rt->tile_col.end = imin(tile_col_end4, rf->iw4);
688
254k
}
689
690
static void load_tmvs_c(const refmvs_frame *const rf, int tile_row_idx,
691
                        const int col_start8, const int col_end8,
692
                        const int row_start8, int row_end8)
693
8.39k
{
694
8.39k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
695
8.39k
    assert(row_start8 >= 0);
696
8.39k
    assert((unsigned) (row_end8 - row_start8) <= 16U);
697
8.39k
    row_end8 = imin(row_end8, rf->ih8);
698
8.39k
    const int col_start8i = imax(col_start8 - 8, 0);
699
8.39k
    const int col_end8i = imin(col_end8 + 8, rf->iw8);
700
701
8.39k
    const ptrdiff_t stride = rf->rp_stride;
702
8.39k
    refmvs_temporal_block *rp_proj =
703
8.39k
        &rf->rp_proj[16 * stride * tile_row_idx + (row_start8 & 15) * stride];
704
55.2k
    for (int y = row_start8; y < row_end8; y++) {
705
272k
        for (int x = col_start8; x < col_end8; x++)
706
226k
            rp_proj[x].mv.n = INVALID_MV;
707
46.8k
        rp_proj += stride;
708
46.8k
    }
709
710
8.39k
    rp_proj = &rf->rp_proj[16 * stride * tile_row_idx];
711
16.2k
    for (int n = 0; n < rf->n_mfmvs; n++) {
712
7.89k
        const int ref2cur = rf->mfmv_ref2cur[n];
713
7.89k
        if (ref2cur == INVALID_REF2CUR) continue;
714
715
7.25k
        const int ref = rf->mfmv_ref[n];
716
7.25k
        const int ref_sign = ref - 4;
717
7.25k
        const refmvs_temporal_block *r = &rf->rp_ref[ref][row_start8 * stride];
718
33.2k
        for (int y = row_start8; y < row_end8; y++) {
719
26.0k
            const int y_sb_align = y & ~7;
720
26.0k
            const int y_proj_start = imax(y_sb_align, row_start8);
721
26.0k
            const int y_proj_end = imin(y_sb_align + 8, row_end8);
722
74.1k
            for (int x = col_start8i; x < col_end8i; x++) {
723
48.1k
                const refmvs_temporal_block *rb = &r[x];
724
48.1k
                const int b_ref = rb->ref;
725
48.1k
                if (!b_ref) continue;
726
18.2k
                const int ref2ref = rf->mfmv_ref2ref[n][b_ref - 1];
727
18.2k
                if (!ref2ref) continue;
728
16.2k
                const mv b_mv = rb->mv;
729
16.2k
                const mv offset = mv_projection(b_mv, ref2cur, ref2ref);
730
16.2k
                int pos_x = x + apply_sign(abs(offset.x) >> 6,
731
16.2k
                                           offset.x ^ ref_sign);
732
16.2k
                const int pos_y = y + apply_sign(abs(offset.y) >> 6,
733
16.2k
                                                 offset.y ^ ref_sign);
734
16.2k
                if (pos_y >= y_proj_start && pos_y < y_proj_end) {
735
15.1k
                    const ptrdiff_t pos = (pos_y & 15) * stride;
736
52.7k
                    for (;;) {
737
52.7k
                        const int x_sb_align = x & ~7;
738
52.7k
                        if (pos_x >= imax(x_sb_align - 8, col_start8) &&
739
52.4k
                            pos_x < imin(x_sb_align + 16, col_end8))
740
51.5k
                        {
741
51.5k
                            rp_proj[pos + pos_x].mv = rb->mv;
742
51.5k
                            rp_proj[pos + pos_x].ref = ref2ref;
743
51.5k
                        }
744
52.7k
                        if (++x >= col_end8i) break;
745
38.7k
                        rb++;
746
38.7k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
747
37.5k
                        pos_x++;
748
37.5k
                    }
749
15.1k
                } else {
750
3.52k
                    for (;;) {
751
3.52k
                        if (++x >= col_end8i) break;
752
2.67k
                        rb++;
753
2.67k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
754
2.67k
                    }
755
1.03k
                }
756
16.2k
                x--;
757
16.2k
            }
758
26.0k
            r += stride;
759
26.0k
        }
760
7.25k
    }
761
8.39k
}
762
763
static void save_tmvs_c(refmvs_temporal_block *rp, const ptrdiff_t stride,
764
                        refmvs_block *const *const rr,
765
                        const uint8_t *const ref_sign,
766
                        const int col_end8, const int row_end8,
767
                        const int col_start8, const int row_start8)
768
39.1k
{
769
267k
    for (int y = row_start8; y < row_end8; y++) {
770
228k
        const refmvs_block *const b = rr[(y & 15) * 2];
771
772
537k
        for (int x = col_start8; x < col_end8;) {
773
309k
            const refmvs_block *const cand_b = &b[x * 2 + 1];
774
309k
            const int bw8 = (dav1d_block_dimensions[cand_b->bs][0] + 1) >> 1;
775
776
309k
            if (cand_b->ref.ref[1] > 0 && ref_sign[cand_b->ref.ref[1] - 1] &&
777
38.7k
                (abs(cand_b->mv.mv[1].y) | abs(cand_b->mv.mv[1].x)) < 4096)
778
33.0k
            {
779
33.0k
                const refmvs_temporal_block tmv = {
780
33.0k
                    .mv = cand_b->mv.mv[1],
781
33.0k
                    .ref = cand_b->ref.ref[1],
782
33.0k
                };
783
127k
                for (int n = 0; n < bw8; n++, x++)
784
94.3k
                    rp[x] = tmv;
785
276k
            } else if (cand_b->ref.ref[0] > 0 && ref_sign[cand_b->ref.ref[0] - 1] &&
786
95.1k
                       (abs(cand_b->mv.mv[0].y) | abs(cand_b->mv.mv[0].x)) < 4096)
787
90.4k
            {
788
90.4k
                const refmvs_temporal_block tmv = {
789
90.4k
                    .mv = cand_b->mv.mv[0],
790
90.4k
                    .ref = cand_b->ref.ref[0],
791
90.4k
                };
792
333k
                for (int n = 0; n < bw8; n++, x++)
793
242k
                    rp[x] = tmv;
794
186k
            } else {
795
186k
                const refmvs_temporal_block tmv = { .mv = { .n = 0 }, .ref = 0 };
796
743k
                for (int n = 0; n < bw8; n++, x++)
797
557k
                    rp[x] = tmv;
798
186k
            }
799
309k
        }
800
228k
        rp += stride;
801
228k
    }
802
39.1k
}
803
804
int dav1d_refmvs_init_frame(refmvs_frame *const rf,
805
                            const Dav1dSequenceHeader *const seq_hdr,
806
                            const Dav1dFrameHeader *const frm_hdr,
807
                            const uint8_t ref_poc[7],
808
                            refmvs_temporal_block *const rp,
809
                            const uint8_t ref_ref_poc[7][7],
810
                            /*const*/ refmvs_temporal_block *const rp_ref[7],
811
                            const int n_tile_threads, const int n_frame_threads)
812
93.6k
{
813
93.6k
    const int rp_stride = ((frm_hdr->width[0] + 127) & ~127) >> 3;
814
18.4E
    const int n_tile_rows = n_tile_threads > 1 ? frm_hdr->tiling.rows : 1;
815
93.6k
    const int n_blocks = rp_stride * n_tile_rows;
816
817
93.6k
    rf->sbsz = 16 << seq_hdr->sb128;
818
93.6k
    rf->frm_hdr = frm_hdr;
819
93.6k
    rf->iw8 = (frm_hdr->width[0] + 7) >> 3;
820
93.6k
    rf->ih8 = (frm_hdr->height + 7) >> 3;
821
93.6k
    rf->iw4 = rf->iw8 << 1;
822
93.6k
    rf->ih4 = rf->ih8 << 1;
823
93.6k
    rf->rp = rp;
824
93.6k
    rf->rp_stride = rp_stride;
825
93.6k
    rf->n_tile_threads = n_tile_threads;
826
93.6k
    rf->n_frame_threads = n_frame_threads;
827
828
93.6k
    if (n_blocks != rf->n_blocks) {
829
23.6k
        const size_t r_sz = sizeof(*rf->r) * 35 * 2 * n_blocks * (1 + (n_frame_threads > 1));
830
23.6k
        const size_t rp_proj_sz = sizeof(*rf->rp_proj) * 16 * n_blocks;
831
        /* Note that sizeof(*rf->r) == 12, but it's accessed using 16-byte unaligned
832
         * loads in save_tmvs() asm which can overread 4 bytes into rp_proj. */
833
23.6k
        dav1d_free_aligned(rf->r);
834
23.6k
        rf->r = dav1d_alloc_aligned(ALLOC_REFMVS, r_sz + rp_proj_sz, 64);
835
23.6k
        if (!rf->r) {
836
0
            rf->n_blocks = 0;
837
0
            return DAV1D_ERR(ENOMEM);
838
0
        }
839
840
23.6k
        rf->rp_proj = (refmvs_temporal_block*)((uintptr_t)rf->r + r_sz);
841
23.6k
        rf->n_blocks = n_blocks;
842
23.6k
    }
843
844
93.6k
    const int poc = frm_hdr->frame_offset;
845
749k
    for (int i = 0; i < 7; i++) {
846
655k
        const int poc_diff = get_poc_diff(seq_hdr->order_hint_n_bits,
847
655k
                                          ref_poc[i], poc);
848
655k
        rf->sign_bias[i] = poc_diff > 0;
849
655k
        rf->mfmv_sign[i] = poc_diff < 0;
850
655k
        rf->pocdiff[i] = iclip(get_poc_diff(seq_hdr->order_hint_n_bits,
851
655k
                                            poc, ref_poc[i]), -31, 31);
852
655k
    }
853
854
    // temporal MV setup
855
93.6k
    rf->n_mfmvs = 0;
856
93.6k
    rf->rp_ref = rp_ref;
857
93.6k
    if (frm_hdr->use_ref_frame_mvs && seq_hdr->order_hint_n_bits) {
858
6.21k
        int total = 2;
859
6.21k
        if (rp_ref[0] && ref_ref_poc[0][6] != ref_poc[3] /* alt-of-last != gold */) {
860
1.81k
            rf->mfmv_ref[rf->n_mfmvs++] = 0; // last
861
1.81k
            total = 3;
862
1.81k
        }
863
6.21k
        if (rp_ref[4] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[4],
864
4.09k
                                      frm_hdr->frame_offset) > 0)
865
360
        {
866
360
            rf->mfmv_ref[rf->n_mfmvs++] = 4; // bwd
867
360
        }
868
6.21k
        if (rp_ref[5] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[5],
869
4.55k
                                      frm_hdr->frame_offset) > 0)
870
227
        {
871
227
            rf->mfmv_ref[rf->n_mfmvs++] = 5; // altref2
872
227
        }
873
6.21k
        if (rf->n_mfmvs < total && rp_ref[6] &&
874
2.50k
            get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[6],
875
2.50k
                         frm_hdr->frame_offset) > 0)
876
517
        {
877
517
            rf->mfmv_ref[rf->n_mfmvs++] = 6; // altref
878
517
        }
879
6.21k
        if (rf->n_mfmvs < total && rp_ref[1])
880
3.90k
            rf->mfmv_ref[rf->n_mfmvs++] = 1; // last2
881
882
13.0k
        for (int n = 0; n < rf->n_mfmvs; n++) {
883
6.82k
            const int rpoc = ref_poc[rf->mfmv_ref[n]];
884
6.82k
            const int diff1 = get_poc_diff(seq_hdr->order_hint_n_bits,
885
6.82k
                                           rpoc, frm_hdr->frame_offset);
886
6.82k
            if (abs(diff1) > 31) {
887
654
                rf->mfmv_ref2cur[n] = INVALID_REF2CUR;
888
6.17k
            } else {
889
6.17k
                rf->mfmv_ref2cur[n] = rf->mfmv_ref[n] < 4 ? -diff1 : diff1;
890
49.3k
                for (int m = 0; m < 7; m++) {
891
43.1k
                    const int rrpoc = ref_ref_poc[rf->mfmv_ref[n]][m];
892
43.1k
                    const int diff2 = get_poc_diff(seq_hdr->order_hint_n_bits,
893
43.1k
                                                   rpoc, rrpoc);
894
                    // unsigned comparison also catches the < 0 case
895
43.1k
                    rf->mfmv_ref2ref[n][m] = (unsigned) diff2 > 31U ? 0 : diff2;
896
43.1k
                }
897
6.17k
            }
898
6.82k
        }
899
6.21k
    }
900
93.6k
    rf->use_ref_frame_mvs = rf->n_mfmvs > 0;
901
902
93.6k
    return 0;
903
93.6k
}
904
905
static void splat_mv_c(refmvs_block **rr, const refmvs_block *const rmv,
906
                       const int bx4, const int bw4, int bh4)
907
959k
{
908
3.31M
    do {
909
3.31M
        refmvs_block *const r = *rr++ + bx4;
910
25.5M
        for (int x = 0; x < bw4; x++)
911
22.2M
            r[x] = *rmv;
912
3.31M
    } while (--bh4);
913
959k
}
914
915
#if HAVE_ASM
916
#if ARCH_AARCH64 || ARCH_ARM
917
#include "src/arm/refmvs.h"
918
#elif ARCH_LOONGARCH64
919
#include "src/loongarch/refmvs.h"
920
#elif ARCH_X86
921
#include "src/x86/refmvs.h"
922
#endif
923
#endif
924
925
COLD void dav1d_refmvs_dsp_init(Dav1dRefmvsDSPContext *const c)
926
28.3k
{
927
28.3k
    c->load_tmvs = load_tmvs_c;
928
28.3k
    c->save_tmvs = save_tmvs_c;
929
28.3k
    c->splat_mv = splat_mv_c;
930
931
#if HAVE_ASM
932
#if ARCH_AARCH64 || ARCH_ARM
933
    refmvs_dsp_init_arm(c);
934
#elif ARCH_LOONGARCH64
935
    refmvs_dsp_init_loongarch(c);
936
#elif ARCH_X86
937
    refmvs_dsp_init_x86(c);
938
#endif
939
#endif
940
28.3k
}