Coverage Report

Created: 2026-08-13 07:20

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/dav1d/src/refmvs.c
Line
Count
Source
1
/*
2
 * Copyright © 2020, VideoLAN and dav1d authors
3
 * Copyright © 2020, Two Orioles, LLC
4
 * All rights reserved.
5
 *
6
 * Redistribution and use in source and binary forms, with or without
7
 * modification, are permitted provided that the following conditions are met:
8
 *
9
 * 1. Redistributions of source code must retain the above copyright notice, this
10
 *    list of conditions and the following disclaimer.
11
 *
12
 * 2. Redistributions in binary form must reproduce the above copyright notice,
13
 *    this list of conditions and the following disclaimer in the documentation
14
 *    and/or other materials provided with the distribution.
15
 *
16
 * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
17
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
18
 * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
19
 * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
20
 * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
21
 * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
22
 * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
23
 * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
24
 * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
25
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
26
 */
27
28
#include "config.h"
29
30
#include <limits.h>
31
#include <stdlib.h>
32
33
#include "dav1d/common.h"
34
35
#include "common/intops.h"
36
37
#include "src/env.h"
38
#include "src/mem.h"
39
#include "src/refmvs.h"
40
41
static void add_spatial_candidate(refmvs_candidate *const mvstack, int *const cnt,
42
                                  const int weight, const refmvs_block *const b,
43
                                  const union refmvs_refpair ref, const mv gmv[2],
44
                                  int *const have_newmv_match,
45
                                  int *const have_refmv_match)
46
2.68M
{
47
2.68M
    if (b->mv.mv[0].n == INVALID_MV) return; // intra block, no intrabc
48
49
2.52M
    if (ref.ref[1] == -1) {
50
2.97M
        for (int n = 0; n < 2; n++) {
51
2.64M
            if (b->ref.ref[n] == ref.ref[0]) {
52
1.94M
                const mv cand_mv = ((b->mf & 1) && gmv[0].n != INVALID_MV) ?
53
1.88M
                                   gmv[0] : b->mv.mv[n];
54
55
1.94M
                *have_refmv_match = 1;
56
1.94M
                *have_newmv_match |= b->mf >> 1;
57
58
1.94M
                const int last = *cnt;
59
3.42M
                for (int m = 0; m < last; m++)
60
2.11M
                    if (mvstack[m].mv.mv[0].n == cand_mv.n) {
61
642k
                        mvstack[m].weight += weight;
62
642k
                        return;
63
642k
                    }
64
65
1.30M
                if (last < 8) {
66
1.30M
                    mvstack[last].mv.mv[0] = cand_mv;
67
1.30M
                    mvstack[last].weight = weight;
68
1.30M
                    *cnt = last + 1;
69
1.30M
                }
70
1.30M
                return;
71
1.94M
            }
72
2.64M
        }
73
2.27M
    } else if (b->ref.pair == ref.pair) {
74
85.3k
        const refmvs_mvpair cand_mv = { .mv = {
75
85.3k
            [0] = ((b->mf & 1) && gmv[0].n != INVALID_MV) ? gmv[0] : b->mv.mv[0],
76
85.3k
            [1] = ((b->mf & 1) && gmv[1].n != INVALID_MV) ? gmv[1] : b->mv.mv[1],
77
85.3k
        }};
78
79
85.3k
        *have_refmv_match = 1;
80
85.3k
        *have_newmv_match |= b->mf >> 1;
81
82
85.3k
        const int last = *cnt;
83
125k
        for (int n = 0; n < last; n++)
84
71.3k
            if (mvstack[n].mv.n == cand_mv.n) {
85
30.8k
                mvstack[n].weight += weight;
86
30.8k
                return;
87
30.8k
            }
88
89
54.4k
        if (last < 8) {
90
54.4k
            mvstack[last].mv = cand_mv;
91
54.4k
            mvstack[last].weight = weight;
92
54.4k
            *cnt = last + 1;
93
54.4k
        }
94
54.4k
    }
95
2.52M
}
96
97
static int scan_row(refmvs_candidate *const mvstack, int *const cnt,
98
                    const union refmvs_refpair ref, const mv gmv[2],
99
                    const refmvs_block *b, const int bw4, const int w4,
100
                    const int max_rows, const int step,
101
                    int *const have_newmv_match, int *const have_refmv_match)
102
778k
{
103
778k
    const refmvs_block *cand_b = b;
104
778k
    const enum BlockSize first_cand_bs = cand_b->bs;
105
778k
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
106
778k
    int cand_bw4 = first_cand_b_dim[0];
107
778k
    int len = imax(step, imin(bw4, cand_bw4));
108
109
778k
    if (bw4 <= cand_bw4) {
110
        // FIXME weight can be higher for odd blocks (bx4 & 1), but then the
111
        // position of the first block has to be odd already, i.e. not just
112
        // for row_offset=-3/-5
113
        // FIXME why can this not be cand_bw4?
114
693k
        const int weight = bw4 == 1 ? 2 :
115
693k
                           imax(2, imin(2 * max_rows, first_cand_b_dim[1]));
116
693k
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
117
693k
                              have_newmv_match, have_refmv_match);
118
693k
        return weight >> 1;
119
693k
    }
120
121
150k
    for (int x = 0;;) {
122
        // FIXME if we overhang above, we could fill a bitmask so we don't have
123
        // to repeat the add_spatial_candidate() for the next row, but just increase
124
        // the weight here
125
150k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
126
150k
                              have_newmv_match, have_refmv_match);
127
150k
        x += len;
128
150k
        if (x >= w4) return 1;
129
66.6k
        cand_b = &b[x];
130
66.6k
        cand_bw4 = dav1d_block_dimensions[cand_b->bs][0];
131
66.6k
        assert(cand_bw4 < bw4);
132
66.6k
        len = imax(step, cand_bw4);
133
66.6k
    }
134
84.8k
}
135
136
static int scan_col(refmvs_candidate *const mvstack, int *const cnt,
137
                    const union refmvs_refpair ref, const mv gmv[2],
138
                    /*const*/ refmvs_block *const *b, const int bh4, const int h4,
139
                    const int bx4, const int max_cols, const int step,
140
                    int *const have_newmv_match, int *const have_refmv_match)
141
1.20M
{
142
1.20M
    const refmvs_block *cand_b = &b[0][bx4];
143
1.20M
    const enum BlockSize first_cand_bs = cand_b->bs;
144
1.20M
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
145
1.20M
    int cand_bh4 = first_cand_b_dim[1];
146
1.20M
    int len = imax(step, imin(bh4, cand_bh4));
147
148
1.20M
    if (bh4 <= cand_bh4) {
149
        // FIXME weight can be higher for odd blocks (by4 & 1), but then the
150
        // position of the first block has to be odd already, i.e. not just
151
        // for col_offset=-3/-5
152
        // FIXME why can this not be cand_bh4?
153
1.09M
        const int weight = bh4 == 1 ? 2 :
154
1.09M
                           imax(2, imin(2 * max_cols, first_cand_b_dim[0]));
155
1.09M
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
156
1.09M
                            have_newmv_match, have_refmv_match);
157
1.09M
        return weight >> 1;
158
1.09M
    }
159
160
196k
    for (int y = 0;;) {
161
        // FIXME if we overhang above, we could fill a bitmask so we don't have
162
        // to repeat the add_spatial_candidate() for the next row, but just increase
163
        // the weight here
164
196k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
165
196k
                              have_newmv_match, have_refmv_match);
166
196k
        y += len;
167
196k
        if (y >= h4) return 1;
168
94.9k
        cand_b = &b[y][bx4];
169
94.9k
        cand_bh4 = dav1d_block_dimensions[cand_b->bs][1];
170
94.9k
        assert(cand_bh4 < bh4);
171
94.9k
        len = imax(step, cand_bh4);
172
94.9k
    }
173
104k
}
174
175
87.6k
static inline union mv mv_projection(const union mv mv, const int num, const int den) {
176
87.6k
    static const uint16_t div_mult[32] = {
177
87.6k
           0, 16384, 8192, 5461, 4096, 3276, 2730, 2340,
178
87.6k
        2048,  1820, 1638, 1489, 1365, 1260, 1170, 1092,
179
87.6k
        1024,   963,  910,  862,  819,  780,  744,  712,
180
87.6k
         682,   655,  630,  606,  585,  564,  546,  528
181
87.6k
    };
182
87.6k
    assert(den > 0 && den < 32);
183
87.6k
    assert(num > -32 && num < 32);
184
87.6k
    const int frac = num * div_mult[den];
185
87.6k
    const int y = mv.y * frac, x = mv.x * frac;
186
    // Round and clip according to AV1 spec section 7.9.3
187
87.6k
    return (union mv) { // 0x3fff == (1 << 14) - 1
188
87.6k
        .y = iclip((y + 8192 + (y >> 31)) >> 14, -0x3fff, 0x3fff),
189
87.6k
        .x = iclip((x + 8192 + (x >> 31)) >> 14, -0x3fff, 0x3fff)
190
87.6k
    };
191
87.6k
}
192
193
static void add_temporal_candidate(const refmvs_frame *const rf,
194
                                   refmvs_candidate *const mvstack, int *const cnt,
195
                                   const refmvs_temporal_block *const rb,
196
                                   const union refmvs_refpair ref, int *const globalmv_ctx,
197
                                   const union mv gmv[])
198
84.9k
{
199
84.9k
    if (rb->mv.n == INVALID_MV) return;
200
201
52.2k
    union mv mv = mv_projection(rb->mv, rf->pocdiff[ref.ref[0] - 1], rb->ref);
202
52.2k
    fix_mv_precision(rf->frm_hdr, &mv);
203
204
52.2k
    const int last = *cnt;
205
52.2k
    if (ref.ref[1] == -1) {
206
34.0k
        if (globalmv_ctx)
207
9.23k
            *globalmv_ctx = (abs(mv.x - gmv[0].x) | abs(mv.y - gmv[0].y)) >= 16;
208
209
57.2k
        for (int n = 0; n < last; n++)
210
49.7k
            if (mvstack[n].mv.mv[0].n == mv.n) {
211
26.5k
                mvstack[n].weight += 2;
212
26.5k
                return;
213
26.5k
            }
214
7.57k
        if (last < 8) {
215
7.49k
            mvstack[last].mv.mv[0] = mv;
216
7.49k
            mvstack[last].weight = 2;
217
7.49k
            *cnt = last + 1;
218
7.49k
        }
219
18.2k
    } else {
220
18.2k
        refmvs_mvpair mvp = { .mv = {
221
18.2k
            [0] = mv,
222
18.2k
            [1] = mv_projection(rb->mv, rf->pocdiff[ref.ref[1] - 1], rb->ref),
223
18.2k
        }};
224
18.2k
        fix_mv_precision(rf->frm_hdr, &mvp.mv[1]);
225
226
29.2k
        for (int n = 0; n < last; n++)
227
25.7k
            if (mvstack[n].mv.n == mvp.n) {
228
14.7k
                mvstack[n].weight += 2;
229
14.7k
                return;
230
14.7k
            }
231
3.44k
        if (last < 8) {
232
3.40k
            mvstack[last].mv = mvp;
233
3.40k
            mvstack[last].weight = 2;
234
3.40k
            *cnt = last + 1;
235
3.40k
        }
236
3.44k
    }
237
52.2k
}
238
239
static void add_compound_extended_candidate(refmvs_candidate *const same,
240
                                            int *const same_count,
241
                                            const refmvs_block *const cand_b,
242
                                            const int sign0, const int sign1,
243
                                            const union refmvs_refpair ref,
244
                                            const uint8_t *const sign_bias)
245
69.2k
{
246
69.2k
    refmvs_candidate *const diff = &same[2];
247
69.2k
    int *const diff_count = &same_count[2];
248
249
175k
    for (int n = 0; n < 2; n++) {
250
135k
        const int cand_ref = cand_b->ref.ref[n];
251
252
135k
        if (cand_ref <= 0) break;
253
254
106k
        mv cand_mv = cand_b->mv.mv[n];
255
106k
        if (cand_ref == ref.ref[0]) {
256
38.7k
            if (same_count[0] < 2)
257
37.6k
                same[same_count[0]++].mv.mv[0] = cand_mv;
258
38.7k
            if (diff_count[1] < 2) {
259
33.9k
                if (sign1 ^ sign_bias[cand_ref - 1]) {
260
2.42k
                    cand_mv.y = -cand_mv.y;
261
2.42k
                    cand_mv.x = -cand_mv.x;
262
2.42k
                }
263
33.9k
                diff[diff_count[1]++].mv.mv[1] = cand_mv;
264
33.9k
            }
265
67.6k
        } else if (cand_ref == ref.ref[1]) {
266
38.4k
            if (same_count[1] < 2)
267
37.7k
                same[same_count[1]++].mv.mv[1] = cand_mv;
268
38.4k
            if (diff_count[0] < 2) {
269
32.8k
                if (sign0 ^ sign_bias[cand_ref - 1]) {
270
2.89k
                    cand_mv.y = -cand_mv.y;
271
2.89k
                    cand_mv.x = -cand_mv.x;
272
2.89k
                }
273
32.8k
                diff[diff_count[0]++].mv.mv[0] = cand_mv;
274
32.8k
            }
275
38.4k
        } else {
276
29.2k
            mv i_cand_mv = (union mv) {
277
29.2k
                .x = -cand_mv.x,
278
29.2k
                .y = -cand_mv.y
279
29.2k
            };
280
281
29.2k
            if (diff_count[0] < 2) {
282
23.4k
                diff[diff_count[0]++].mv.mv[0] =
283
23.4k
                    sign0 ^ sign_bias[cand_ref - 1] ?
284
22.6k
                    i_cand_mv : cand_mv;
285
23.4k
            }
286
287
29.2k
            if (diff_count[1] < 2) {
288
22.2k
                diff[diff_count[1]++].mv.mv[1] =
289
22.2k
                    sign1 ^ sign_bias[cand_ref - 1] ?
290
21.5k
                    i_cand_mv : cand_mv;
291
22.2k
            }
292
29.2k
        }
293
106k
    }
294
69.2k
}
295
296
static void add_single_extended_candidate(refmvs_candidate mvstack[8], int *const cnt,
297
                                          const refmvs_block *const cand_b,
298
                                          const int sign, const uint8_t *const sign_bias)
299
345k
{
300
694k
    for (int n = 0; n < 2; n++) {
301
681k
        const int cand_ref = cand_b->ref.ref[n];
302
303
681k
        if (cand_ref <= 0) break;
304
        // we need to continue even if cand_ref == ref.ref[0], since
305
        // the candidate could have been added as a globalmv variant,
306
        // which changes the value
307
        // FIXME if scan_{row,col}() returned a mask for the nearest
308
        // edge, we could skip the appropriate ones here
309
310
348k
        mv cand_mv = cand_b->mv.mv[n];
311
348k
        if (sign ^ sign_bias[cand_ref - 1]) {
312
6.04k
            cand_mv.y = -cand_mv.y;
313
6.04k
            cand_mv.x = -cand_mv.x;
314
6.04k
        }
315
316
348k
        int m;
317
348k
        const int last = *cnt;
318
382k
        for (m = 0; m < last; m++)
319
319k
            if (cand_mv.n == mvstack[m].mv.mv[0].n)
320
285k
                break;
321
348k
        if (m == last) {
322
62.8k
            mvstack[m].mv.mv[0] = cand_mv;
323
62.8k
            mvstack[m].weight = 2; // "minimal"
324
62.8k
            *cnt = last + 1;
325
62.8k
        }
326
348k
    }
327
345k
}
328
329
/*
330
 * refmvs_frame allocates memory for one sbrow (32 blocks high, whole frame
331
 * wide) of 4x4-resolution refmvs_block entries for spatial MV referencing.
332
 * mvrefs_tile[] keeps a list of 35 (32 + 3 above) pointers into this memory,
333
 * and each sbrow, the bottom entries (y=27/29/31) are exchanged with the top
334
 * (-5/-3/-1) pointers by calling dav1d_refmvs_tile_sbrow_init() at the start
335
 * of each tile/sbrow.
336
 *
337
 * For temporal MV referencing, we call dav1d_refmvs_save_tmvs() at the end of
338
 * each tile/sbrow (when tile column threading is enabled), or at the start of
339
 * each interleaved sbrow (i.e. once for all tile columns together, when tile
340
 * column threading is disabled). This will copy the 4x4-resolution spatial MVs
341
 * into 8x8-resolution refmvs_temporal_block structures. Then, for subsequent
342
 * frames, at the start of each tile/sbrow (when tile column threading is
343
 * enabled) or at the start of each interleaved sbrow (when tile column
344
 * threading is disabled), we call load_tmvs(), which will project the MVs to
345
 * their respective position in the current frame.
346
 */
347
348
void dav1d_refmvs_find(const refmvs_tile *const rt,
349
                       refmvs_candidate mvstack[8], int *const cnt,
350
                       int *const ctx,
351
                       const union refmvs_refpair ref, const enum BlockSize bs,
352
                       const enum EdgeFlags edge_flags,
353
                       const int by4, const int bx4)
354
763k
{
355
763k
    const refmvs_frame *const rf = rt->rf;
356
763k
    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
357
763k
    const int bw4 = b_dim[0], w4 = imin(imin(bw4, 16), rt->tile_col.end - bx4);
358
763k
    const int bh4 = b_dim[1], h4 = imin(imin(bh4, 16), rt->tile_row.end - by4);
359
763k
    mv gmv[2], tgmv[2];
360
361
763k
    *cnt = 0;
362
763k
    assert(ref.ref[0] >=  0 && ref.ref[0] <= 8 &&
363
763k
           ref.ref[1] >= -1 && ref.ref[1] <= 8);
364
763k
    if (ref.ref[0] > 0) {
365
440k
        tgmv[0] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[0] - 1],
366
440k
                             bx4, by4, bw4, bh4, rf->frm_hdr);
367
440k
        gmv[0] = rf->frm_hdr->gmv[ref.ref[0] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
368
344k
                 tgmv[0] : (mv) { .n = INVALID_MV };
369
440k
    } else {
370
322k
        tgmv[0] = (mv) { .n = 0 };
371
322k
        gmv[0] = (mv) { .n = INVALID_MV };
372
322k
    }
373
763k
    if (ref.ref[1] > 0) {
374
73.4k
        tgmv[1] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[1] - 1],
375
73.4k
                             bx4, by4, bw4, bh4, rf->frm_hdr);
376
73.4k
        gmv[1] = rf->frm_hdr->gmv[ref.ref[1] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
377
54.2k
                 tgmv[1] : (mv) { .n = INVALID_MV };
378
73.4k
    }
379
380
    // top
381
763k
    int have_newmv = 0, have_col_mvs = 0, have_row_mvs = 0;
382
763k
    unsigned max_rows = 0, n_rows = ~0;
383
763k
    const refmvs_block *b_top;
384
763k
    if (by4 > rt->tile_row.start) {
385
462k
        max_rows = imin((by4 - rt->tile_row.start + 1) >> 1, 2 + (bh4 > 1));
386
462k
        b_top = &rt->r[(by4 & 31) - 1 + 5][bx4];
387
462k
        n_rows = scan_row(mvstack, cnt, ref, gmv, b_top,
388
462k
                          bw4, w4, max_rows, bw4 >= 16 ? 4 : 1,
389
462k
                          &have_newmv, &have_row_mvs);
390
462k
    }
391
392
    // left
393
763k
    unsigned max_cols = 0, n_cols = ~0U;
394
763k
    refmvs_block *const *b_left;
395
763k
    if (bx4 > rt->tile_col.start) {
396
582k
        max_cols = imin((bx4 - rt->tile_col.start + 1) >> 1, 2 + (bw4 > 1));
397
582k
        b_left = &rt->r[(by4 & 31) + 5];
398
582k
        n_cols = scan_col(mvstack, cnt, ref, gmv, b_left,
399
582k
                          bh4, h4, bx4 - 1, max_cols, bh4 >= 16 ? 4 : 1,
400
582k
                          &have_newmv, &have_col_mvs);
401
582k
    }
402
403
    // top/right
404
763k
    if (n_rows != ~0U && edge_flags & EDGE_I444_TOP_HAS_RIGHT &&
405
266k
        imax(bw4, bh4) <= 16 && bw4 + bx4 < rt->tile_col.end)
406
188k
    {
407
188k
        add_spatial_candidate(mvstack, cnt, 4, &b_top[bw4], ref, gmv,
408
188k
                              &have_newmv, &have_row_mvs);
409
188k
    }
410
411
763k
    const int nearest_match = have_col_mvs + have_row_mvs;
412
763k
    const int nearest_cnt = *cnt;
413
1.61M
    for (int n = 0; n < nearest_cnt; n++)
414
852k
        mvstack[n].weight += 640;
415
416
    // temporal
417
763k
    int globalmv_ctx = rf->frm_hdr->use_ref_frame_mvs;
418
763k
    if (rf->use_ref_frame_mvs) {
419
31.9k
        const ptrdiff_t stride = rf->rp_stride;
420
31.9k
        const int by8 = by4 >> 1, bx8 = bx4 >> 1;
421
31.9k
        const refmvs_temporal_block *const rbi = &rt->rp_proj[(by8 & 15) * stride + bx8];
422
31.9k
        const refmvs_temporal_block *rb = rbi;
423
31.9k
        const int step_h = bw4 >= 16 ? 2 : 1, step_v = bh4 >= 16 ? 2 : 1;
424
31.9k
        const int w8 = imin((w4 + 1) >> 1, 8), h8 = imin((h4 + 1) >> 1, 8);
425
82.9k
        for (int y = 0; y < h8; y += step_v) {
426
123k
            for (int x = 0; x < w8; x+= step_h) {
427
72.1k
                add_temporal_candidate(rf, mvstack, cnt, &rb[x], ref,
428
72.1k
                                       !(x | y) ? &globalmv_ctx : NULL, tgmv);
429
72.1k
            }
430
51.0k
            rb += stride * step_v;
431
51.0k
        }
432
31.9k
        if (imin(bw4, bh4) >= 2 && imax(bw4, bh4) < 16) {
433
14.2k
            const int bh8 = bh4 >> 1, bw8 = bw4 >> 1;
434
14.2k
            rb = &rbi[bh8 * stride];
435
14.2k
            const int has_bottom = by8 + bh8 < imin(rt->tile_row.end >> 1,
436
14.2k
                                                    (by8 & ~7) + 8);
437
14.2k
            if (has_bottom && bx8 - 1 >= imax(rt->tile_col.start >> 1, bx8 & ~7)) {
438
3.76k
                add_temporal_candidate(rf, mvstack, cnt, &rb[-1], ref,
439
3.76k
                                       NULL, NULL);
440
3.76k
            }
441
14.2k
            if (bx8 + bw8 < imin(rt->tile_col.end >> 1, (bx8 & ~7) + 8)) {
442
6.42k
                if (has_bottom) {
443
3.62k
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8], ref,
444
3.62k
                                           NULL, NULL);
445
3.62k
                }
446
6.42k
                if (by8 + bh8 - 1 < imin(rt->tile_row.end >> 1, (by8 & ~7) + 8)) {
447
5.40k
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8 - stride],
448
5.40k
                                           ref, NULL, NULL);
449
5.40k
                }
450
6.42k
            }
451
14.2k
        }
452
31.9k
    }
453
763k
    assert(*cnt <= 8);
454
455
    // top/left (which, confusingly, is part of "secondary" references)
456
763k
    int have_dummy_newmv_match;
457
763k
    if ((n_rows | n_cols) != ~0U) {
458
361k
        add_spatial_candidate(mvstack, cnt, 4, &b_top[-1], ref, gmv,
459
361k
                              &have_dummy_newmv_match, &have_row_mvs);
460
361k
    }
461
462
    // "secondary" (non-direct neighbour) top & left edges
463
    // what is different about secondary is that everything is now in 8x8 resolution
464
2.28M
    for (int n = 2; n <= 3; n++) {
465
1.52M
        if ((unsigned) n > n_rows && (unsigned) n <= max_rows) {
466
316k
            n_rows += scan_row(mvstack, cnt, ref, gmv,
467
316k
                               &rt->r[(((by4 & 31) - 2 * n + 1) | 1) + 5][bx4 | 1],
468
316k
                               bw4, w4, 1 + max_rows - n, bw4 >= 16 ? 4 : 2,
469
316k
                               &have_dummy_newmv_match, &have_row_mvs);
470
316k
        }
471
472
1.52M
        if ((unsigned) n > n_cols && (unsigned) n <= max_cols) {
473
620k
            n_cols += scan_col(mvstack, cnt, ref, gmv, &rt->r[((by4 & 31) | 1) + 5],
474
620k
                               bh4, h4, (bx4 - n * 2 + 1) | 1,
475
620k
                               1 + max_cols - n, bh4 >= 16 ? 4 : 2,
476
620k
                               &have_dummy_newmv_match, &have_col_mvs);
477
620k
        }
478
1.52M
    }
479
763k
    assert(*cnt <= 8);
480
481
763k
    const int ref_match_count = have_col_mvs + have_row_mvs;
482
483
    // context build-up
484
763k
    int refmv_ctx, newmv_ctx;
485
763k
    switch (nearest_match) {
486
170k
    case 0:
487
170k
        refmv_ctx = imin(2, ref_match_count);
488
170k
        newmv_ctx = ref_match_count > 0;
489
170k
        break;
490
331k
    case 1:
491
331k
        refmv_ctx = imin(ref_match_count * 3, 4);
492
331k
        newmv_ctx = 3 - have_newmv;
493
331k
        break;
494
261k
    case 2:
495
261k
        refmv_ctx = 5;
496
261k
        newmv_ctx = 5 - have_newmv;
497
261k
        break;
498
763k
    }
499
500
    // sorting (nearest, then "secondary")
501
763k
    int len = nearest_cnt;
502
1.50M
    while (len) {
503
737k
        int last = 0;
504
1.06M
        for (int n = 1; n < len; n++) {
505
323k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
506
157k
#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
507
148k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
508
148k
                last = n;
509
148k
            }
510
323k
        }
511
737k
        len = last;
512
737k
    }
513
763k
    len = *cnt;
514
1.13M
    while (len > nearest_cnt) {
515
373k
        int last = nearest_cnt;
516
536k
        for (int n = nearest_cnt + 1; n < len; n++) {
517
163k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
518
9.19k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
519
9.19k
#undef EXCHANGE
520
9.19k
                last = n;
521
9.19k
            }
522
163k
        }
523
373k
        len = last;
524
373k
    }
525
526
763k
    if (ref.ref[1] > 0) {
527
73.4k
        if (*cnt < 2) {
528
59.6k
            const int sign0 = rf->sign_bias[ref.ref[0] - 1];
529
59.6k
            const int sign1 = rf->sign_bias[ref.ref[1] - 1];
530
59.6k
            const int sz4 = imin(w4, h4);
531
59.6k
            refmvs_candidate *const same = &mvstack[*cnt];
532
59.6k
            int same_count[4] = { 0 };
533
534
            // non-self references in top
535
67.6k
            if (n_rows != ~0U) for (int x = 0; x < sz4;) {
536
35.0k
                const refmvs_block *const cand_b = &b_top[x];
537
35.0k
                add_compound_extended_candidate(same, same_count, cand_b,
538
35.0k
                                                sign0, sign1, ref, rf->sign_bias);
539
35.0k
                x += dav1d_block_dimensions[cand_b->bs][0];
540
35.0k
            }
541
542
            // non-self references in left
543
64.7k
            if (n_cols != ~0U) for (int y = 0; y < sz4;) {
544
34.2k
                const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
545
34.2k
                add_compound_extended_candidate(same, same_count, cand_b,
546
34.2k
                                                sign0, sign1, ref, rf->sign_bias);
547
34.2k
                y += dav1d_block_dimensions[cand_b->bs][1];
548
34.2k
            }
549
550
59.6k
            refmvs_candidate *const diff = &same[2];
551
59.6k
            const int *const diff_count = &same_count[2];
552
553
            // merge together
554
178k
            for (int n = 0; n < 2; n++) {
555
119k
                int m = same_count[n];
556
557
119k
                if (m >= 2) continue;
558
559
102k
                const int l = diff_count[n];
560
102k
                if (l) {
561
57.3k
                    same[m].mv.mv[n] = diff[0].mv.mv[n];
562
57.3k
                    if (++m == 2) continue;
563
21.5k
                    if (l == 2) {
564
13.7k
                        same[1].mv.mv[n] = diff[1].mv.mv[n];
565
13.7k
                        continue;
566
13.7k
                    }
567
21.5k
                }
568
92.0k
                do {
569
92.0k
                    same[m].mv.mv[n] = tgmv[n];
570
92.0k
                } while (++m < 2);
571
52.8k
            }
572
573
            // if the first extended was the same as the non-extended one,
574
            // then replace it with the second extended one
575
59.6k
            int n = *cnt;
576
59.6k
            if (n == 1 && mvstack[0].mv.n == same[0].mv.n)
577
16.1k
                mvstack[1].mv = mvstack[2].mv;
578
97.7k
            do {
579
97.7k
                mvstack[n].weight = 2;
580
97.7k
            } while (++n < 2);
581
59.6k
            *cnt = 2;
582
59.6k
        }
583
584
        // clamping
585
73.4k
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
586
73.4k
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
587
73.4k
        const int top = -(by4 + bh4 + 4) * 4 * 8;
588
73.4k
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
589
590
73.4k
        const int n_refmvs = *cnt;
591
73.4k
        int n = 0;
592
155k
        do {
593
155k
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
594
155k
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
595
155k
            mvstack[n].mv.mv[1].x = iclip(mvstack[n].mv.mv[1].x, left, right);
596
155k
            mvstack[n].mv.mv[1].y = iclip(mvstack[n].mv.mv[1].y, top, bottom);
597
155k
        } while (++n < n_refmvs);
598
599
73.4k
        switch (refmv_ctx >> 1) {
600
41.7k
        case 0:
601
41.7k
            *ctx = imin(newmv_ctx, 1);
602
41.7k
            break;
603
21.2k
        case 1:
604
21.2k
            *ctx = 1 + imin(newmv_ctx, 3);
605
21.2k
            break;
606
10.4k
        case 2:
607
10.4k
            *ctx = iclip(3 + newmv_ctx, 4, 7);
608
10.4k
            break;
609
73.4k
        }
610
611
73.4k
        return;
612
689k
    } else if (*cnt < 2 && ref.ref[0] > 0) {
613
286k
        const int sign = rf->sign_bias[ref.ref[0] - 1];
614
286k
        const int sz4 = imin(w4, h4);
615
616
        // non-self references in top
617
359k
        if (n_rows != ~0U) for (int x = 0; x < sz4 && *cnt < 2;) {
618
183k
            const refmvs_block *const cand_b = &b_top[x];
619
183k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
620
183k
            x += dav1d_block_dimensions[cand_b->bs][0];
621
183k
        }
622
623
        // non-self references in left
624
325k
        if (n_cols != ~0U) for (int y = 0; y < sz4 && *cnt < 2;) {
625
162k
            const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
626
162k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
627
162k
            y += dav1d_block_dimensions[cand_b->bs][1];
628
162k
        }
629
286k
    }
630
689k
    assert(*cnt <= 8);
631
632
    // clamping
633
689k
    int n_refmvs = *cnt;
634
689k
    if (n_refmvs) {
635
607k
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
636
607k
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
637
607k
        const int top = -(by4 + bh4 + 4) * 4 * 8;
638
607k
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
639
640
607k
        int n = 0;
641
1.38M
        do {
642
1.38M
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
643
1.38M
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
644
1.38M
        } while (++n < n_refmvs);
645
607k
    }
646
647
1.07M
    for (int n = *cnt; n < 2; n++)
648
380k
        mvstack[n].mv.mv[0] = tgmv[0];
649
650
689k
    *ctx = (refmv_ctx << 4) | (globalmv_ctx << 3) | newmv_ctx;
651
689k
}
652
653
void dav1d_refmvs_tile_sbrow_init(refmvs_tile *const rt, const refmvs_frame *const rf,
654
                                  const int tile_col_start4, const int tile_col_end4,
655
                                  const int tile_row_start4, const int tile_row_end4,
656
                                  const int sby, int tile_row_idx, const int pass)
657
255k
{
658
255k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
659
255k
    rt->rp_proj = &rf->rp_proj[16 * rf->rp_stride * tile_row_idx];
660
255k
    const ptrdiff_t r_stride = rf->rp_stride * 2;
661
255k
    const ptrdiff_t pass_off = (rf->n_frame_threads > 1 && pass == 2) ?
662
132k
        35 * 2 * rf->n_blocks : 0;
663
255k
    refmvs_block *r = &rf->r[35 * r_stride * tile_row_idx + pass_off];
664
255k
    const int sbsz = rf->sbsz;
665
255k
    const int off = (sbsz * sby) & 16;
666
5.27M
    for (int i = 0; i < sbsz; i++, r += r_stride)
667
5.01M
        rt->r[off + 5 + i] = r;
668
255k
    rt->r[off + 0] = r;
669
255k
    r += r_stride;
670
255k
    rt->r[off + 1] = NULL;
671
255k
    rt->r[off + 2] = r;
672
255k
    r += r_stride;
673
255k
    rt->r[off + 3] = NULL;
674
255k
    rt->r[off + 4] = r;
675
255k
    if (sby & 1) {
676
151k
#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
677
50.5k
        EXCHANGE(rt->r[off + 0], rt->r[off + sbsz + 0]);
678
50.5k
        EXCHANGE(rt->r[off + 2], rt->r[off + sbsz + 2]);
679
50.5k
        EXCHANGE(rt->r[off + 4], rt->r[off + sbsz + 4]);
680
50.5k
#undef EXCHANGE
681
50.5k
    }
682
683
255k
    rt->rf = rf;
684
255k
    rt->tile_row.start = tile_row_start4;
685
255k
    rt->tile_row.end = imin(tile_row_end4, rf->ih4);
686
255k
    rt->tile_col.start = tile_col_start4;
687
255k
    rt->tile_col.end = imin(tile_col_end4, rf->iw4);
688
255k
}
689
690
static void load_tmvs_c(const refmvs_frame *const rf, int tile_row_idx,
691
                        const int col_start8, const int col_end8,
692
                        const int row_start8, int row_end8)
693
17.1k
{
694
17.1k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
695
17.1k
    assert(row_start8 >= 0);
696
17.1k
    assert((unsigned) (row_end8 - row_start8) <= 16U);
697
17.1k
    row_end8 = imin(row_end8, rf->ih8);
698
17.1k
    const int col_start8i = imax(col_start8 - 8, 0);
699
17.1k
    const int col_end8i = imin(col_end8 + 8, rf->iw8);
700
701
17.1k
    const ptrdiff_t stride = rf->rp_stride;
702
17.1k
    refmvs_temporal_block *rp_proj =
703
17.1k
        &rf->rp_proj[16 * stride * tile_row_idx + (row_start8 & 15) * stride];
704
79.0k
    for (int y = row_start8; y < row_end8; y++) {
705
322k
        for (int x = col_start8; x < col_end8; x++)
706
260k
            rp_proj[x].mv.n = INVALID_MV;
707
61.9k
        rp_proj += stride;
708
61.9k
    }
709
710
17.1k
    rp_proj = &rf->rp_proj[16 * stride * tile_row_idx];
711
39.7k
    for (int n = 0; n < rf->n_mfmvs; n++) {
712
22.5k
        const int ref2cur = rf->mfmv_ref2cur[n];
713
22.5k
        if (ref2cur == INVALID_REF2CUR) continue;
714
715
13.0k
        const int ref = rf->mfmv_ref[n];
716
13.0k
        const int ref_sign = ref - 4;
717
13.0k
        const refmvs_temporal_block *r = &rf->rp_ref[ref][row_start8 * stride];
718
50.8k
        for (int y = row_start8; y < row_end8; y++) {
719
37.7k
            const int y_sb_align = y & ~7;
720
37.7k
            const int y_proj_start = imax(y_sb_align, row_start8);
721
37.7k
            const int y_proj_end = imin(y_sb_align + 8, row_end8);
722
92.4k
            for (int x = col_start8i; x < col_end8i; x++) {
723
54.7k
                const refmvs_temporal_block *rb = &r[x];
724
54.7k
                const int b_ref = rb->ref;
725
54.7k
                if (!b_ref) continue;
726
21.5k
                const int ref2ref = rf->mfmv_ref2ref[n][b_ref - 1];
727
21.5k
                if (!ref2ref) continue;
728
17.1k
                const mv b_mv = rb->mv;
729
17.1k
                const mv offset = mv_projection(b_mv, ref2cur, ref2ref);
730
17.1k
                int pos_x = x + apply_sign(abs(offset.x) >> 6,
731
17.1k
                                           offset.x ^ ref_sign);
732
17.1k
                const int pos_y = y + apply_sign(abs(offset.y) >> 6,
733
17.1k
                                                 offset.y ^ ref_sign);
734
17.1k
                if (pos_y >= y_proj_start && pos_y < y_proj_end) {
735
15.9k
                    const ptrdiff_t pos = (pos_y & 15) * stride;
736
66.4k
                    for (;;) {
737
66.4k
                        const int x_sb_align = x & ~7;
738
66.4k
                        if (pos_x >= imax(x_sb_align - 8, col_start8) &&
739
65.1k
                            pos_x < imin(x_sb_align + 16, col_end8))
740
64.2k
                        {
741
64.2k
                            rp_proj[pos + pos_x].mv = rb->mv;
742
64.2k
                            rp_proj[pos + pos_x].ref = ref2ref;
743
64.2k
                        }
744
66.4k
                        if (++x >= col_end8i) break;
745
52.7k
                        rb++;
746
52.7k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
747
50.4k
                        pos_x++;
748
50.4k
                    }
749
15.9k
                } else {
750
4.65k
                    for (;;) {
751
4.65k
                        if (++x >= col_end8i) break;
752
3.94k
                        rb++;
753
3.94k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
754
3.94k
                    }
755
1.16k
                }
756
17.1k
                x--;
757
17.1k
            }
758
37.7k
            r += stride;
759
37.7k
        }
760
13.0k
    }
761
17.1k
}
762
763
static void save_tmvs_c(refmvs_temporal_block *rp, const ptrdiff_t stride,
764
                        refmvs_block *const *const rr,
765
                        const uint8_t *const ref_sign,
766
                        const int col_end8, const int row_end8,
767
                        const int col_start8, const int row_start8)
768
47.4k
{
769
304k
    for (int y = row_start8; y < row_end8; y++) {
770
256k
        const refmvs_block *const b = rr[(y & 15) * 2];
771
772
606k
        for (int x = col_start8; x < col_end8;) {
773
349k
            const refmvs_block *const cand_b = &b[x * 2 + 1];
774
349k
            const int bw8 = (dav1d_block_dimensions[cand_b->bs][0] + 1) >> 1;
775
776
349k
            if (cand_b->ref.ref[1] > 0 && ref_sign[cand_b->ref.ref[1] - 1] &&
777
43.8k
                (abs(cand_b->mv.mv[1].y) | abs(cand_b->mv.mv[1].x)) < 4096)
778
38.0k
            {
779
38.0k
                const refmvs_temporal_block tmv = {
780
38.0k
                    .mv = cand_b->mv.mv[1],
781
38.0k
                    .ref = cand_b->ref.ref[1],
782
38.0k
                };
783
152k
                for (int n = 0; n < bw8; n++, x++)
784
114k
                    rp[x] = tmv;
785
311k
            } else if (cand_b->ref.ref[0] > 0 && ref_sign[cand_b->ref.ref[0] - 1] &&
786
102k
                       (abs(cand_b->mv.mv[0].y) | abs(cand_b->mv.mv[0].x)) < 4096)
787
98.3k
            {
788
98.3k
                const refmvs_temporal_block tmv = {
789
98.3k
                    .mv = cand_b->mv.mv[0],
790
98.3k
                    .ref = cand_b->ref.ref[0],
791
98.3k
                };
792
356k
                for (int n = 0; n < bw8; n++, x++)
793
258k
                    rp[x] = tmv;
794
213k
            } else {
795
213k
                const refmvs_temporal_block tmv = { .mv = { .n = 0 }, .ref = 0 };
796
865k
                for (int n = 0; n < bw8; n++, x++)
797
651k
                    rp[x] = tmv;
798
213k
            }
799
349k
        }
800
256k
        rp += stride;
801
256k
    }
802
47.4k
}
803
804
int dav1d_refmvs_init_frame(refmvs_frame *const rf,
805
                            const Dav1dSequenceHeader *const seq_hdr,
806
                            const Dav1dFrameHeader *const frm_hdr,
807
                            const uint8_t ref_poc[7],
808
                            refmvs_temporal_block *const rp,
809
                            const uint8_t ref_ref_poc[7][7],
810
                            /*const*/ refmvs_temporal_block *const rp_ref[7],
811
                            const int n_tile_threads, const int n_frame_threads)
812
95.3k
{
813
95.3k
    const int rp_stride = ((frm_hdr->width[0] + 127) & ~127) >> 3;
814
95.3k
    const int n_tile_rows = n_tile_threads > 1 ? frm_hdr->tiling.rows : 1;
815
95.3k
    const int n_blocks = rp_stride * n_tile_rows;
816
817
95.3k
    rf->sbsz = 16 << seq_hdr->sb128;
818
95.3k
    rf->frm_hdr = frm_hdr;
819
95.3k
    rf->iw8 = (frm_hdr->width[0] + 7) >> 3;
820
95.3k
    rf->ih8 = (frm_hdr->height + 7) >> 3;
821
95.3k
    rf->iw4 = rf->iw8 << 1;
822
95.3k
    rf->ih4 = rf->ih8 << 1;
823
95.3k
    rf->rp = rp;
824
95.3k
    rf->rp_stride = rp_stride;
825
95.3k
    rf->n_tile_threads = n_tile_threads;
826
95.3k
    rf->n_frame_threads = n_frame_threads;
827
828
95.3k
    if (n_blocks != rf->n_blocks) {
829
24.0k
        const size_t r_sz = sizeof(*rf->r) * 35 * 2 * n_blocks * (1 + (n_frame_threads > 1));
830
24.0k
        const size_t rp_proj_sz = sizeof(*rf->rp_proj) * 16 * n_blocks;
831
        /* Note that sizeof(*rf->r) == 12, but it's accessed using 16-byte unaligned
832
         * loads in save_tmvs() asm which can overread 4 bytes into rp_proj. */
833
24.0k
        dav1d_free_aligned(rf->r);
834
24.0k
        rf->r = dav1d_alloc_aligned(ALLOC_REFMVS, r_sz + rp_proj_sz, 64);
835
24.0k
        if (!rf->r) {
836
0
            rf->n_blocks = 0;
837
0
            return DAV1D_ERR(ENOMEM);
838
0
        }
839
840
24.0k
        rf->rp_proj = (refmvs_temporal_block*)((uintptr_t)rf->r + r_sz);
841
24.0k
        rf->n_blocks = n_blocks;
842
24.0k
    }
843
844
95.3k
    const int poc = frm_hdr->frame_offset;
845
762k
    for (int i = 0; i < 7; i++) {
846
667k
        const int poc_diff = get_poc_diff(seq_hdr->order_hint_n_bits,
847
667k
                                          ref_poc[i], poc);
848
667k
        rf->sign_bias[i] = poc_diff > 0;
849
667k
        rf->mfmv_sign[i] = poc_diff < 0;
850
667k
        rf->pocdiff[i] = iclip(get_poc_diff(seq_hdr->order_hint_n_bits,
851
667k
                                            poc, ref_poc[i]), -31, 31);
852
667k
    }
853
854
    // temporal MV setup
855
95.3k
    rf->n_mfmvs = 0;
856
95.3k
    rf->rp_ref = rp_ref;
857
95.3k
    if (frm_hdr->use_ref_frame_mvs && seq_hdr->order_hint_n_bits) {
858
14.7k
        int total = 2;
859
14.7k
        if (rp_ref[0] && ref_ref_poc[0][6] != ref_poc[3] /* alt-of-last != gold */) {
860
6.36k
            rf->mfmv_ref[rf->n_mfmvs++] = 0; // last
861
6.36k
            total = 3;
862
6.36k
        }
863
14.7k
        if (rp_ref[4] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[4],
864
12.3k
                                      frm_hdr->frame_offset) > 0)
865
701
        {
866
701
            rf->mfmv_ref[rf->n_mfmvs++] = 4; // bwd
867
701
        }
868
14.7k
        if (rp_ref[5] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[5],
869
12.3k
                                      frm_hdr->frame_offset) > 0)
870
622
        {
871
622
            rf->mfmv_ref[rf->n_mfmvs++] = 5; // altref2
872
622
        }
873
14.7k
        if (rf->n_mfmvs < total && rp_ref[6] &&
874
8.99k
            get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[6],
875
8.99k
                         frm_hdr->frame_offset) > 0)
876
4.59k
        {
877
4.59k
            rf->mfmv_ref[rf->n_mfmvs++] = 6; // altref
878
4.59k
        }
879
14.7k
        if (rf->n_mfmvs < total && rp_ref[1])
880
9.37k
            rf->mfmv_ref[rf->n_mfmvs++] = 1; // last2
881
882
36.4k
        for (int n = 0; n < rf->n_mfmvs; n++) {
883
21.6k
            const int rpoc = ref_poc[rf->mfmv_ref[n]];
884
21.6k
            const int diff1 = get_poc_diff(seq_hdr->order_hint_n_bits,
885
21.6k
                                           rpoc, frm_hdr->frame_offset);
886
21.6k
            if (abs(diff1) > 31) {
887
9.55k
                rf->mfmv_ref2cur[n] = INVALID_REF2CUR;
888
12.1k
            } else {
889
12.1k
                rf->mfmv_ref2cur[n] = rf->mfmv_ref[n] < 4 ? -diff1 : diff1;
890
96.8k
                for (int m = 0; m < 7; m++) {
891
84.7k
                    const int rrpoc = ref_ref_poc[rf->mfmv_ref[n]][m];
892
84.7k
                    const int diff2 = get_poc_diff(seq_hdr->order_hint_n_bits,
893
84.7k
                                                   rpoc, rrpoc);
894
                    // unsigned comparison also catches the < 0 case
895
84.7k
                    rf->mfmv_ref2ref[n][m] = (unsigned) diff2 > 31U ? 0 : diff2;
896
84.7k
                }
897
12.1k
            }
898
21.6k
        }
899
14.7k
    }
900
95.3k
    rf->use_ref_frame_mvs = rf->n_mfmvs > 0;
901
902
95.3k
    return 0;
903
95.3k
}
904
905
static void splat_mv_c(refmvs_block **rr, const refmvs_block *const rmv,
906
                       const int bx4, const int bw4, int bh4)
907
1.03M
{
908
3.42M
    do {
909
3.42M
        refmvs_block *const r = *rr++ + bx4;
910
26.6M
        for (int x = 0; x < bw4; x++)
911
23.1M
            r[x] = *rmv;
912
3.42M
    } while (--bh4);
913
1.03M
}
914
915
#if HAVE_ASM
916
#if ARCH_AARCH64 || ARCH_ARM
917
#include "src/arm/refmvs.h"
918
#elif ARCH_LOONGARCH64
919
#include "src/loongarch/refmvs.h"
920
#elif ARCH_X86
921
#include "src/x86/refmvs.h"
922
#endif
923
#endif
924
925
COLD void dav1d_refmvs_dsp_init(Dav1dRefmvsDSPContext *const c)
926
30.1k
{
927
30.1k
    c->load_tmvs = load_tmvs_c;
928
30.1k
    c->save_tmvs = save_tmvs_c;
929
30.1k
    c->splat_mv = splat_mv_c;
930
931
#if HAVE_ASM
932
#if ARCH_AARCH64 || ARCH_ARM
933
    refmvs_dsp_init_arm(c);
934
#elif ARCH_LOONGARCH64
935
    refmvs_dsp_init_loongarch(c);
936
#elif ARCH_X86
937
    refmvs_dsp_init_x86(c);
938
#endif
939
#endif
940
30.1k
}