Coverage Report

Created: 2026-09-14 06:44

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/dav1d/src/refmvs.c
Line
Count
Source
1
/*
2
 * Copyright © 2020, VideoLAN and dav1d authors
3
 * Copyright © 2020, Two Orioles, LLC
4
 * All rights reserved.
5
 *
6
 * Redistribution and use in source and binary forms, with or without
7
 * modification, are permitted provided that the following conditions are met:
8
 *
9
 * 1. Redistributions of source code must retain the above copyright notice, this
10
 *    list of conditions and the following disclaimer.
11
 *
12
 * 2. Redistributions in binary form must reproduce the above copyright notice,
13
 *    this list of conditions and the following disclaimer in the documentation
14
 *    and/or other materials provided with the distribution.
15
 *
16
 * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
17
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
18
 * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
19
 * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
20
 * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
21
 * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
22
 * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
23
 * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
24
 * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
25
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
26
 */
27
28
#include "config.h"
29
30
#include <limits.h>
31
#include <stdlib.h>
32
33
#include "dav1d/common.h"
34
35
#include "common/intops.h"
36
37
#include "src/env.h"
38
#include "src/mem.h"
39
#include "src/refmvs.h"
40
41
static void add_spatial_candidate(refmvs_candidate *const mvstack, int *const cnt,
42
                                  const int weight, const refmvs_block *const b,
43
                                  const union refmvs_refpair ref, const mv gmv[2],
44
                                  int *const have_newmv_match,
45
                                  int *const have_refmv_match)
46
2.59M
{
47
2.59M
    if (b->mv.mv[0].n == INVALID_MV) return; // intra block, no intrabc
48
49
2.16M
    if (ref.ref[1] == -1) {
50
2.42M
        for (int n = 0; n < 2; n++) {
51
2.13M
            if (b->ref.ref[n] == ref.ref[0]) {
52
1.49M
                const mv cand_mv = ((b->mf & 1) && gmv[0].n != INVALID_MV) ?
53
1.46M
                                   gmv[0] : b->mv.mv[n];
54
55
1.49M
                *have_refmv_match = 1;
56
1.49M
                *have_newmv_match |= b->mf >> 1;
57
58
1.49M
                const int last = *cnt;
59
2.94M
                for (int m = 0; m < last; m++)
60
1.95M
                    if (mvstack[m].mv.mv[0].n == cand_mv.n) {
61
511k
                        mvstack[m].weight += weight;
62
511k
                        return;
63
511k
                    }
64
65
983k
                if (last < 8) {
66
981k
                    mvstack[last].mv.mv[0] = cand_mv;
67
981k
                    mvstack[last].weight = weight;
68
981k
                    *cnt = last + 1;
69
981k
                }
70
983k
                return;
71
1.49M
            }
72
2.13M
        }
73
1.78M
    } else if (b->ref.pair == ref.pair) {
74
116k
        const refmvs_mvpair cand_mv = { .mv = {
75
116k
            [0] = ((b->mf & 1) && gmv[0].n != INVALID_MV) ? gmv[0] : b->mv.mv[0],
76
116k
            [1] = ((b->mf & 1) && gmv[1].n != INVALID_MV) ? gmv[1] : b->mv.mv[1],
77
116k
        }};
78
79
116k
        *have_refmv_match = 1;
80
116k
        *have_newmv_match |= b->mf >> 1;
81
82
116k
        const int last = *cnt;
83
171k
        for (int n = 0; n < last; n++)
84
99.4k
            if (mvstack[n].mv.n == cand_mv.n) {
85
44.5k
                mvstack[n].weight += weight;
86
44.5k
                return;
87
44.5k
            }
88
89
72.1k
        if (last < 8) {
90
72.0k
            mvstack[last].mv = cand_mv;
91
72.0k
            mvstack[last].weight = weight;
92
72.0k
            *cnt = last + 1;
93
72.0k
        }
94
72.1k
    }
95
2.16M
}
96
97
static int scan_row(refmvs_candidate *const mvstack, int *const cnt,
98
                    const union refmvs_refpair ref, const mv gmv[2],
99
                    const refmvs_block *b, const int bw4, const int w4,
100
                    const int max_rows, const int step,
101
                    int *const have_newmv_match, int *const have_refmv_match)
102
809k
{
103
809k
    const refmvs_block *cand_b = b;
104
809k
    const enum BlockSize first_cand_bs = cand_b->bs;
105
809k
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
106
809k
    int cand_bw4 = first_cand_b_dim[0];
107
809k
    int len = imax(step, imin(bw4, cand_bw4));
108
109
809k
    if (bw4 <= cand_bw4) {
110
        // FIXME weight can be higher for odd blocks (bx4 & 1), but then the
111
        // position of the first block has to be odd already, i.e. not just
112
        // for row_offset=-3/-5
113
        // FIXME why can this not be cand_bw4?
114
708k
        const int weight = bw4 == 1 ? 2 :
115
708k
                           imax(2, imin(2 * max_rows, first_cand_b_dim[1]));
116
708k
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
117
708k
                              have_newmv_match, have_refmv_match);
118
708k
        return weight >> 1;
119
708k
    }
120
121
204k
    for (int x = 0;;) {
122
        // FIXME if we overhang above, we could fill a bitmask so we don't have
123
        // to repeat the add_spatial_candidate() for the next row, but just increase
124
        // the weight here
125
204k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
126
204k
                              have_newmv_match, have_refmv_match);
127
204k
        x += len;
128
204k
        if (x >= w4) return 1;
129
102k
        cand_b = &b[x];
130
102k
        cand_bw4 = dav1d_block_dimensions[cand_b->bs][0];
131
102k
        assert(cand_bw4 < bw4);
132
102k
        len = imax(step, cand_bw4);
133
102k
    }
134
101k
}
135
136
static int scan_col(refmvs_candidate *const mvstack, int *const cnt,
137
                    const union refmvs_refpair ref, const mv gmv[2],
138
                    /*const*/ refmvs_block *const *b, const int bh4, const int h4,
139
                    const int bx4, const int max_cols, const int step,
140
                    int *const have_newmv_match, int *const have_refmv_match)
141
980k
{
142
980k
    const refmvs_block *cand_b = &b[0][bx4];
143
980k
    const enum BlockSize first_cand_bs = cand_b->bs;
144
980k
    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
145
980k
    int cand_bh4 = first_cand_b_dim[1];
146
980k
    int len = imax(step, imin(bh4, cand_bh4));
147
148
980k
    if (bh4 <= cand_bh4) {
149
        // FIXME weight can be higher for odd blocks (by4 & 1), but then the
150
        // position of the first block has to be odd already, i.e. not just
151
        // for col_offset=-3/-5
152
        // FIXME why can this not be cand_bh4?
153
852k
        const int weight = bh4 == 1 ? 2 :
154
852k
                           imax(2, imin(2 * max_cols, first_cand_b_dim[0]));
155
852k
        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
156
852k
                            have_newmv_match, have_refmv_match);
157
852k
        return weight >> 1;
158
852k
    }
159
160
253k
    for (int y = 0;;) {
161
        // FIXME if we overhang above, we could fill a bitmask so we don't have
162
        // to repeat the add_spatial_candidate() for the next row, but just increase
163
        // the weight here
164
253k
        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
165
253k
                              have_newmv_match, have_refmv_match);
166
253k
        y += len;
167
253k
        if (y >= h4) return 1;
168
125k
        cand_b = &b[y][bx4];
169
125k
        cand_bh4 = dav1d_block_dimensions[cand_b->bs][1];
170
125k
        assert(cand_bh4 < bh4);
171
125k
        len = imax(step, cand_bh4);
172
125k
    }
173
128k
}
174
175
93.3k
static inline union mv mv_projection(const union mv mv, const int num, const int den) {
176
93.3k
    static const uint16_t div_mult[32] = {
177
93.3k
           0, 16384, 8192, 5461, 4096, 3276, 2730, 2340,
178
93.3k
        2048,  1820, 1638, 1489, 1365, 1260, 1170, 1092,
179
93.3k
        1024,   963,  910,  862,  819,  780,  744,  712,
180
93.3k
         682,   655,  630,  606,  585,  564,  546,  528
181
93.3k
    };
182
93.3k
    assert(den > 0 && den < 32);
183
93.3k
    assert(num > -32 && num < 32);
184
93.3k
    const int frac = num * div_mult[den];
185
93.3k
    const int y = mv.y * frac, x = mv.x * frac;
186
    // Round and clip according to AV1 spec section 7.9.3
187
93.3k
    return (union mv) { // 0x3fff == (1 << 14) - 1
188
93.3k
        .y = iclip((y + 8192 + (y >> 31)) >> 14, -0x3fff, 0x3fff),
189
93.3k
        .x = iclip((x + 8192 + (x >> 31)) >> 14, -0x3fff, 0x3fff)
190
93.3k
    };
191
93.3k
}
192
193
static void add_temporal_candidate(const refmvs_frame *const rf,
194
                                   refmvs_candidate *const mvstack, int *const cnt,
195
                                   const refmvs_temporal_block *const rb,
196
                                   const union refmvs_refpair ref, int *const globalmv_ctx,
197
                                   const union mv gmv[])
198
106k
{
199
106k
    if (rb->mv.n == INVALID_MV) return;
200
201
33.9k
    union mv mv = mv_projection(rb->mv, rf->pocdiff[ref.ref[0] - 1], rb->ref);
202
33.9k
    fix_mv_precision(rf->frm_hdr, &mv);
203
204
33.9k
    const int last = *cnt;
205
33.9k
    if (ref.ref[1] == -1) {
206
21.9k
        if (globalmv_ctx)
207
5.28k
            *globalmv_ctx = (abs(mv.x - gmv[0].x) | abs(mv.y - gmv[0].y)) >= 16;
208
209
26.5k
        for (int n = 0; n < last; n++)
210
21.7k
            if (mvstack[n].mv.mv[0].n == mv.n) {
211
17.2k
                mvstack[n].weight += 2;
212
17.2k
                return;
213
17.2k
            }
214
4.75k
        if (last < 8) {
215
4.75k
            mvstack[last].mv.mv[0] = mv;
216
4.75k
            mvstack[last].weight = 2;
217
4.75k
            *cnt = last + 1;
218
4.75k
        }
219
12.0k
    } else {
220
12.0k
        refmvs_mvpair mvp = { .mv = {
221
12.0k
            [0] = mv,
222
12.0k
            [1] = mv_projection(rb->mv, rf->pocdiff[ref.ref[1] - 1], rb->ref),
223
12.0k
        }};
224
12.0k
        fix_mv_precision(rf->frm_hdr, &mvp.mv[1]);
225
226
15.0k
        for (int n = 0; n < last; n++)
227
11.5k
            if (mvstack[n].mv.n == mvp.n) {
228
8.53k
                mvstack[n].weight += 2;
229
8.53k
                return;
230
8.53k
            }
231
3.48k
        if (last < 8) {
232
3.48k
            mvstack[last].mv = mvp;
233
3.48k
            mvstack[last].weight = 2;
234
3.48k
            *cnt = last + 1;
235
3.48k
        }
236
3.48k
    }
237
33.9k
}
238
239
static void add_compound_extended_candidate(refmvs_candidate *const same,
240
                                            int *const same_count,
241
                                            const refmvs_block *const cand_b,
242
                                            const int sign0, const int sign1,
243
                                            const union refmvs_refpair ref,
244
                                            const uint8_t *const sign_bias)
245
96.4k
{
246
96.4k
    refmvs_candidate *const diff = &same[2];
247
96.4k
    int *const diff_count = &same_count[2];
248
249
248k
    for (int n = 0; n < 2; n++) {
250
190k
        const int cand_ref = cand_b->ref.ref[n];
251
252
190k
        if (cand_ref <= 0) break;
253
254
152k
        mv cand_mv = cand_b->mv.mv[n];
255
152k
        if (cand_ref == ref.ref[0]) {
256
54.2k
            if (same_count[0] < 2)
257
52.4k
                same[same_count[0]++].mv.mv[0] = cand_mv;
258
54.2k
            if (diff_count[1] < 2) {
259
46.0k
                if (sign1 ^ sign_bias[cand_ref - 1]) {
260
2.71k
                    cand_mv.y = -cand_mv.y;
261
2.71k
                    cand_mv.x = -cand_mv.x;
262
2.71k
                }
263
46.0k
                diff[diff_count[1]++].mv.mv[1] = cand_mv;
264
46.0k
            }
265
98.0k
        } else if (cand_ref == ref.ref[1]) {
266
51.7k
            if (same_count[1] < 2)
267
50.4k
                same[same_count[1]++].mv.mv[1] = cand_mv;
268
51.7k
            if (diff_count[0] < 2) {
269
42.4k
                if (sign0 ^ sign_bias[cand_ref - 1]) {
270
2.71k
                    cand_mv.y = -cand_mv.y;
271
2.71k
                    cand_mv.x = -cand_mv.x;
272
2.71k
                }
273
42.4k
                diff[diff_count[0]++].mv.mv[0] = cand_mv;
274
42.4k
            }
275
51.7k
        } else {
276
46.3k
            mv i_cand_mv = (union mv) {
277
46.3k
                .x = -cand_mv.x,
278
46.3k
                .y = -cand_mv.y
279
46.3k
            };
280
281
46.3k
            if (diff_count[0] < 2) {
282
36.6k
                diff[diff_count[0]++].mv.mv[0] =
283
36.6k
                    sign0 ^ sign_bias[cand_ref - 1] ?
284
34.9k
                    i_cand_mv : cand_mv;
285
36.6k
            }
286
287
46.3k
            if (diff_count[1] < 2) {
288
33.8k
                diff[diff_count[1]++].mv.mv[1] =
289
33.8k
                    sign1 ^ sign_bias[cand_ref - 1] ?
290
32.3k
                    i_cand_mv : cand_mv;
291
33.8k
            }
292
46.3k
        }
293
152k
    }
294
96.4k
}
295
296
static void add_single_extended_candidate(refmvs_candidate mvstack[8], int *const cnt,
297
                                          const refmvs_block *const cand_b,
298
                                          const int sign, const uint8_t *const sign_bias)
299
181k
{
300
370k
    for (int n = 0; n < 2; n++) {
301
354k
        const int cand_ref = cand_b->ref.ref[n];
302
303
354k
        if (cand_ref <= 0) break;
304
        // we need to continue even if cand_ref == ref.ref[0], since
305
        // the candidate could have been added as a globalmv variant,
306
        // which changes the value
307
        // FIXME if scan_{row,col}() returned a mask for the nearest
308
        // edge, we could skip the appropriate ones here
309
310
189k
        mv cand_mv = cand_b->mv.mv[n];
311
189k
        if (sign ^ sign_bias[cand_ref - 1]) {
312
4.35k
            cand_mv.y = -cand_mv.y;
313
4.35k
            cand_mv.x = -cand_mv.x;
314
4.35k
        }
315
316
189k
        int m;
317
189k
        const int last = *cnt;
318
224k
        for (m = 0; m < last; m++)
319
171k
            if (cand_mv.n == mvstack[m].mv.mv[0].n)
320
136k
                break;
321
189k
        if (m == last) {
322
52.8k
            mvstack[m].mv.mv[0] = cand_mv;
323
52.8k
            mvstack[m].weight = 2; // "minimal"
324
52.8k
            *cnt = last + 1;
325
52.8k
        }
326
189k
    }
327
181k
}
328
329
/*
330
 * refmvs_frame allocates memory for one sbrow (32 blocks high, whole frame
331
 * wide) of 4x4-resolution refmvs_block entries for spatial MV referencing.
332
 * mvrefs_tile[] keeps a list of 35 (32 + 3 above) pointers into this memory,
333
 * and each sbrow, the bottom entries (y=27/29/31) are exchanged with the top
334
 * (-5/-3/-1) pointers by calling dav1d_refmvs_tile_sbrow_init() at the start
335
 * of each tile/sbrow.
336
 *
337
 * For temporal MV referencing, we call dav1d_refmvs_save_tmvs() at the end of
338
 * each tile/sbrow (when tile column threading is enabled), or at the start of
339
 * each interleaved sbrow (i.e. once for all tile columns together, when tile
340
 * column threading is disabled). This will copy the 4x4-resolution spatial MVs
341
 * into 8x8-resolution refmvs_temporal_block structures. Then, for subsequent
342
 * frames, at the start of each tile/sbrow (when tile column threading is
343
 * enabled) or at the start of each interleaved sbrow (when tile column
344
 * threading is disabled), we call load_tmvs(), which will project the MVs to
345
 * their respective position in the current frame.
346
 */
347
348
void dav1d_refmvs_find(const refmvs_tile *const rt,
349
                       refmvs_candidate mvstack[8], int *const cnt,
350
                       int *const ctx,
351
                       const union refmvs_refpair ref, const enum BlockSize bs,
352
                       const enum EdgeFlags edge_flags,
353
                       const int by4, const int bx4)
354
596k
{
355
596k
    const refmvs_frame *const rf = rt->rf;
356
596k
    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
357
596k
    const int bw4 = b_dim[0], w4 = imin(imin(bw4, 16), rt->tile_col.end - bx4);
358
596k
    const int bh4 = b_dim[1], h4 = imin(imin(bh4, 16), rt->tile_row.end - by4);
359
596k
    mv gmv[2], tgmv[2];
360
361
596k
    *cnt = 0;
362
596k
    assert(ref.ref[0] >=  0 && ref.ref[0] <= 8 &&
363
596k
           ref.ref[1] >= -1 && ref.ref[1] <= 8);
364
596k
    if (ref.ref[0] > 0) {
365
397k
        tgmv[0] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[0] - 1],
366
397k
                             bx4, by4, bw4, bh4, rf->frm_hdr);
367
397k
        gmv[0] = rf->frm_hdr->gmv[ref.ref[0] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
368
264k
                 tgmv[0] : (mv) { .n = INVALID_MV };
369
397k
    } else {
370
198k
        tgmv[0] = (mv) { .n = 0 };
371
198k
        gmv[0] = (mv) { .n = INVALID_MV };
372
198k
    }
373
596k
    if (ref.ref[1] > 0) {
374
95.2k
        tgmv[1] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[1] - 1],
375
95.2k
                             bx4, by4, bw4, bh4, rf->frm_hdr);
376
95.2k
        gmv[1] = rf->frm_hdr->gmv[ref.ref[1] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
377
71.9k
                 tgmv[1] : (mv) { .n = INVALID_MV };
378
95.2k
    }
379
380
    // top
381
596k
    int have_newmv = 0, have_col_mvs = 0, have_row_mvs = 0;
382
596k
    unsigned max_rows = 0, n_rows = ~0;
383
596k
    const refmvs_block *b_top;
384
596k
    if (by4 > rt->tile_row.start) {
385
415k
        max_rows = imin((by4 - rt->tile_row.start + 1) >> 1, 2 + (bh4 > 1));
386
415k
        b_top = &rt->r[(by4 & 31) - 1 + 5][bx4];
387
415k
        n_rows = scan_row(mvstack, cnt, ref, gmv, b_top,
388
415k
                          bw4, w4, max_rows, bw4 >= 16 ? 4 : 1,
389
415k
                          &have_newmv, &have_row_mvs);
390
415k
    }
391
392
    // left
393
596k
    unsigned max_cols = 0, n_cols = ~0U;
394
596k
    refmvs_block *const *b_left;
395
596k
    if (bx4 > rt->tile_col.start) {
396
462k
        max_cols = imin((bx4 - rt->tile_col.start + 1) >> 1, 2 + (bw4 > 1));
397
462k
        b_left = &rt->r[(by4 & 31) + 5];
398
462k
        n_cols = scan_col(mvstack, cnt, ref, gmv, b_left,
399
462k
                          bh4, h4, bx4 - 1, max_cols, bh4 >= 16 ? 4 : 1,
400
462k
                          &have_newmv, &have_col_mvs);
401
462k
    }
402
403
    // top/right
404
596k
    if (n_rows != ~0U && edge_flags & EDGE_I444_TOP_HAS_RIGHT &&
405
242k
        imax(bw4, bh4) <= 16 && bw4 + bx4 < rt->tile_col.end)
406
209k
    {
407
209k
        add_spatial_candidate(mvstack, cnt, 4, &b_top[bw4], ref, gmv,
408
209k
                              &have_newmv, &have_row_mvs);
409
209k
    }
410
411
596k
    const int nearest_match = have_col_mvs + have_row_mvs;
412
596k
    const int nearest_cnt = *cnt;
413
1.24M
    for (int n = 0; n < nearest_cnt; n++)
414
647k
        mvstack[n].weight += 640;
415
416
    // temporal
417
596k
    int globalmv_ctx = rf->frm_hdr->use_ref_frame_mvs;
418
596k
    if (rf->use_ref_frame_mvs) {
419
38.7k
        const ptrdiff_t stride = rf->rp_stride;
420
38.7k
        const int by8 = by4 >> 1, bx8 = bx4 >> 1;
421
38.7k
        const refmvs_temporal_block *const rbi = &rt->rp_proj[(by8 & 15) * stride + bx8];
422
38.7k
        const refmvs_temporal_block *rb = rbi;
423
38.7k
        const int step_h = bw4 >= 16 ? 2 : 1, step_v = bh4 >= 16 ? 2 : 1;
424
38.7k
        const int w8 = imin((w4 + 1) >> 1, 8), h8 = imin((h4 + 1) >> 1, 8);
425
124k
        for (int y = 0; y < h8; y += step_v) {
426
189k
            for (int x = 0; x < w8; x+= step_h) {
427
104k
                add_temporal_candidate(rf, mvstack, cnt, &rb[x], ref,
428
104k
                                       !(x | y) ? &globalmv_ctx : NULL, tgmv);
429
104k
            }
430
85.5k
            rb += stride * step_v;
431
85.5k
        }
432
38.7k
        if (imin(bw4, bh4) >= 2 && imax(bw4, bh4) < 16) {
433
18.4k
            const int bh8 = bh4 >> 1, bw8 = bw4 >> 1;
434
18.4k
            rb = &rbi[bh8 * stride];
435
18.4k
            const int has_bottom = by8 + bh8 < imin(rt->tile_row.end >> 1,
436
18.4k
                                                    (by8 & ~7) + 8);
437
18.4k
            if (has_bottom && bx8 - 1 >= imax(rt->tile_col.start >> 1, bx8 & ~7)) {
438
778
                add_temporal_candidate(rf, mvstack, cnt, &rb[-1], ref,
439
778
                                       NULL, NULL);
440
778
            }
441
18.4k
            if (bx8 + bw8 < imin(rt->tile_col.end >> 1, (bx8 & ~7) + 8)) {
442
1.20k
                if (has_bottom) {
443
791
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8], ref,
444
791
                                           NULL, NULL);
445
791
                }
446
1.20k
                if (by8 + bh8 - 1 < imin(rt->tile_row.end >> 1, (by8 & ~7) + 8)) {
447
1.05k
                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8 - stride],
448
1.05k
                                           ref, NULL, NULL);
449
1.05k
                }
450
1.20k
            }
451
18.4k
        }
452
38.7k
    }
453
596k
    assert(*cnt <= 8);
454
455
    // top/left (which, confusingly, is part of "secondary" references)
456
596k
    int have_dummy_newmv_match;
457
596k
    if ((n_rows | n_cols) != ~0U) {
458
368k
        add_spatial_candidate(mvstack, cnt, 4, &b_top[-1], ref, gmv,
459
368k
                              &have_dummy_newmv_match, &have_row_mvs);
460
368k
    }
461
462
    // "secondary" (non-direct neighbour) top & left edges
463
    // what is different about secondary is that everything is now in 8x8 resolution
464
1.78M
    for (int n = 2; n <= 3; n++) {
465
1.19M
        if ((unsigned) n > n_rows && (unsigned) n <= max_rows) {
466
394k
            n_rows += scan_row(mvstack, cnt, ref, gmv,
467
394k
                               &rt->r[(((by4 & 31) - 2 * n + 1) | 1) + 5][bx4 | 1],
468
394k
                               bw4, w4, 1 + max_rows - n, bw4 >= 16 ? 4 : 2,
469
394k
                               &have_dummy_newmv_match, &have_row_mvs);
470
394k
        }
471
472
1.19M
        if ((unsigned) n > n_cols && (unsigned) n <= max_cols) {
473
519k
            n_cols += scan_col(mvstack, cnt, ref, gmv, &rt->r[((by4 & 31) | 1) + 5],
474
519k
                               bh4, h4, (bx4 - n * 2 + 1) | 1,
475
519k
                               1 + max_cols - n, bh4 >= 16 ? 4 : 2,
476
519k
                               &have_dummy_newmv_match, &have_col_mvs);
477
519k
        }
478
1.19M
    }
479
596k
    assert(*cnt <= 8);
480
481
596k
    const int ref_match_count = have_col_mvs + have_row_mvs;
482
483
    // context build-up
484
596k
    int refmv_ctx, newmv_ctx;
485
596k
    switch (nearest_match) {
486
186k
    case 0:
487
186k
        refmv_ctx = imin(2, ref_match_count);
488
186k
        newmv_ctx = ref_match_count > 0;
489
186k
        break;
490
204k
    case 1:
491
204k
        refmv_ctx = imin(ref_match_count * 3, 4);
492
204k
        newmv_ctx = 3 - have_newmv;
493
204k
        break;
494
207k
    case 2:
495
207k
        refmv_ctx = 5;
496
207k
        newmv_ctx = 5 - have_newmv;
497
207k
        break;
498
596k
    }
499
500
    // sorting (nearest, then "secondary")
501
596k
    int len = nearest_cnt;
502
1.11M
    while (len) {
503
518k
        int last = 0;
504
793k
        for (int n = 1; n < len; n++) {
505
275k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
506
160k
#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
507
115k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
508
115k
                last = n;
509
115k
            }
510
275k
        }
511
518k
        len = last;
512
518k
    }
513
596k
    len = *cnt;
514
898k
    while (len > nearest_cnt) {
515
301k
        int last = nearest_cnt;
516
471k
        for (int n = nearest_cnt + 1; n < len; n++) {
517
169k
            if (mvstack[n - 1].weight < mvstack[n].weight) {
518
45.1k
                EXCHANGE(mvstack[n - 1], mvstack[n]);
519
45.1k
#undef EXCHANGE
520
45.1k
                last = n;
521
45.1k
            }
522
169k
        }
523
301k
        len = last;
524
301k
    }
525
526
596k
    if (ref.ref[1] > 0) {
527
95.2k
        if (*cnt < 2) {
528
76.8k
            const int sign0 = rf->sign_bias[ref.ref[0] - 1];
529
76.8k
            const int sign1 = rf->sign_bias[ref.ref[1] - 1];
530
76.8k
            const int sz4 = imin(w4, h4);
531
76.8k
            refmvs_candidate *const same = &mvstack[*cnt];
532
76.8k
            int same_count[4] = { 0 };
533
534
            // non-self references in top
535
90.7k
            if (n_rows != ~0U) for (int x = 0; x < sz4;) {
536
47.1k
                const refmvs_block *const cand_b = &b_top[x];
537
47.1k
                add_compound_extended_candidate(same, same_count, cand_b,
538
47.1k
                                                sign0, sign1, ref, rf->sign_bias);
539
47.1k
                x += dav1d_block_dimensions[cand_b->bs][0];
540
47.1k
            }
541
542
            // non-self references in left
543
91.9k
            if (n_cols != ~0U) for (int y = 0; y < sz4;) {
544
49.2k
                const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
545
49.2k
                add_compound_extended_candidate(same, same_count, cand_b,
546
49.2k
                                                sign0, sign1, ref, rf->sign_bias);
547
49.2k
                y += dav1d_block_dimensions[cand_b->bs][1];
548
49.2k
            }
549
550
76.8k
            refmvs_candidate *const diff = &same[2];
551
76.8k
            const int *const diff_count = &same_count[2];
552
553
            // merge together
554
230k
            for (int n = 0; n < 2; n++) {
555
153k
                int m = same_count[n];
556
557
153k
                if (m >= 2) continue;
558
559
127k
                const int l = diff_count[n];
560
127k
                if (l) {
561
70.6k
                    same[m].mv.mv[n] = diff[0].mv.mv[n];
562
70.6k
                    if (++m == 2) continue;
563
24.3k
                    if (l == 2) {
564
19.8k
                        same[1].mv.mv[n] = diff[1].mv.mv[n];
565
19.8k
                        continue;
566
19.8k
                    }
567
24.3k
                }
568
114k
                do {
569
114k
                    same[m].mv.mv[n] = tgmv[n];
570
114k
                } while (++m < 2);
571
60.9k
            }
572
573
            // if the first extended was the same as the non-extended one,
574
            // then replace it with the second extended one
575
76.8k
            int n = *cnt;
576
76.8k
            if (n == 1 && mvstack[0].mv.n == same[0].mv.n)
577
19.7k
                mvstack[1].mv = mvstack[2].mv;
578
125k
            do {
579
125k
                mvstack[n].weight = 2;
580
125k
            } while (++n < 2);
581
76.8k
            *cnt = 2;
582
76.8k
        }
583
584
        // clamping
585
95.2k
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
586
95.2k
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
587
95.2k
        const int top = -(by4 + bh4 + 4) * 4 * 8;
588
95.2k
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
589
590
95.2k
        const int n_refmvs = *cnt;
591
95.2k
        int n = 0;
592
200k
        do {
593
200k
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
594
200k
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
595
200k
            mvstack[n].mv.mv[1].x = iclip(mvstack[n].mv.mv[1].x, left, right);
596
200k
            mvstack[n].mv.mv[1].y = iclip(mvstack[n].mv.mv[1].y, top, bottom);
597
200k
        } while (++n < n_refmvs);
598
599
95.2k
        switch (refmv_ctx >> 1) {
600
53.8k
        case 0:
601
53.8k
            *ctx = imin(newmv_ctx, 1);
602
53.8k
            break;
603
24.9k
        case 1:
604
24.9k
            *ctx = 1 + imin(newmv_ctx, 3);
605
24.9k
            break;
606
16.4k
        case 2:
607
16.4k
            *ctx = iclip(3 + newmv_ctx, 4, 7);
608
16.4k
            break;
609
95.2k
        }
610
611
95.2k
        return;
612
501k
    } else if (*cnt < 2 && ref.ref[0] > 0) {
613
181k
        const int sign = rf->sign_bias[ref.ref[0] - 1];
614
181k
        const int sz4 = imin(w4, h4);
615
616
        // non-self references in top
617
187k
        if (n_rows != ~0U) for (int x = 0; x < sz4 && *cnt < 2;) {
618
95.2k
            const refmvs_block *const cand_b = &b_top[x];
619
95.2k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
620
95.2k
            x += dav1d_block_dimensions[cand_b->bs][0];
621
95.2k
        }
622
623
        // non-self references in left
624
182k
        if (n_cols != ~0U) for (int y = 0; y < sz4 && *cnt < 2;) {
625
86.4k
            const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
626
86.4k
            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
627
86.4k
            y += dav1d_block_dimensions[cand_b->bs][1];
628
86.4k
        }
629
181k
    }
630
501k
    assert(*cnt <= 8);
631
632
    // clamping
633
501k
    int n_refmvs = *cnt;
634
501k
    if (n_refmvs) {
635
409k
        const int left = -(bx4 + bw4 + 4) * 4 * 8;
636
409k
        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
637
409k
        const int top = -(by4 + bh4 + 4) * 4 * 8;
638
409k
        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
639
640
409k
        int n = 0;
641
1.04M
        do {
642
1.04M
            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
643
1.04M
            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
644
1.04M
        } while (++n < n_refmvs);
645
409k
    }
646
647
803k
    for (int n = *cnt; n < 2; n++)
648
301k
        mvstack[n].mv.mv[0] = tgmv[0];
649
650
501k
    *ctx = (refmv_ctx << 4) | (globalmv_ctx << 3) | newmv_ctx;
651
501k
}
652
653
void dav1d_refmvs_tile_sbrow_init(refmvs_tile *const rt, const refmvs_frame *const rf,
654
                                  const int tile_col_start4, const int tile_col_end4,
655
                                  const int tile_row_start4, const int tile_row_end4,
656
                                  const int sby, int tile_row_idx, const int pass)
657
230k
{
658
230k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
659
230k
    rt->rp_proj = &rf->rp_proj[16 * rf->rp_stride * tile_row_idx];
660
230k
    const ptrdiff_t r_stride = rf->rp_stride * 2;
661
230k
    const ptrdiff_t pass_off = (rf->n_frame_threads > 1 && pass == 2) ?
662
120k
        35 * 2 * rf->n_blocks : 0;
663
230k
    refmvs_block *r = &rf->r[35 * r_stride * tile_row_idx + pass_off];
664
230k
    const int sbsz = rf->sbsz;
665
230k
    const int off = (sbsz * sby) & 16;
666
5.07M
    for (int i = 0; i < sbsz; i++, r += r_stride)
667
4.84M
        rt->r[off + 5 + i] = r;
668
230k
    rt->r[off + 0] = r;
669
230k
    r += r_stride;
670
230k
    rt->r[off + 1] = NULL;
671
230k
    rt->r[off + 2] = r;
672
230k
    r += r_stride;
673
230k
    rt->r[off + 3] = NULL;
674
230k
    rt->r[off + 4] = r;
675
230k
    if (sby & 1) {
676
75.0k
#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
677
25.0k
        EXCHANGE(rt->r[off + 0], rt->r[off + sbsz + 0]);
678
25.0k
        EXCHANGE(rt->r[off + 2], rt->r[off + sbsz + 2]);
679
25.0k
        EXCHANGE(rt->r[off + 4], rt->r[off + sbsz + 4]);
680
25.0k
#undef EXCHANGE
681
25.0k
    }
682
683
230k
    rt->rf = rf;
684
230k
    rt->tile_row.start = tile_row_start4;
685
230k
    rt->tile_row.end = imin(tile_row_end4, rf->ih4);
686
230k
    rt->tile_col.start = tile_col_start4;
687
230k
    rt->tile_col.end = imin(tile_col_end4, rf->iw4);
688
230k
}
689
690
static void load_tmvs_c(const refmvs_frame *const rf, int tile_row_idx,
691
                        const int col_start8, const int col_end8,
692
                        const int row_start8, int row_end8)
693
20.6k
{
694
20.6k
    if (rf->n_tile_threads == 1) tile_row_idx = 0;
695
20.6k
    assert(row_start8 >= 0);
696
20.6k
    assert((unsigned) (row_end8 - row_start8) <= 16U);
697
20.6k
    row_end8 = imin(row_end8, rf->ih8);
698
20.6k
    const int col_start8i = imax(col_start8 - 8, 0);
699
20.6k
    const int col_end8i = imin(col_end8 + 8, rf->iw8);
700
701
20.6k
    const ptrdiff_t stride = rf->rp_stride;
702
20.6k
    refmvs_temporal_block *rp_proj =
703
20.6k
        &rf->rp_proj[16 * stride * tile_row_idx + (row_start8 & 15) * stride];
704
164k
    for (int y = row_start8; y < row_end8; y++) {
705
318k
        for (int x = col_start8; x < col_end8; x++)
706
175k
            rp_proj[x].mv.n = INVALID_MV;
707
143k
        rp_proj += stride;
708
143k
    }
709
710
20.6k
    rp_proj = &rf->rp_proj[16 * stride * tile_row_idx];
711
47.0k
    for (int n = 0; n < rf->n_mfmvs; n++) {
712
26.4k
        const int ref2cur = rf->mfmv_ref2cur[n];
713
26.4k
        if (ref2cur == INVALID_REF2CUR) continue;
714
715
24.5k
        const int ref = rf->mfmv_ref[n];
716
24.5k
        const int ref_sign = ref - 4;
717
24.5k
        const refmvs_temporal_block *r = &rf->rp_ref[ref][row_start8 * stride];
718
187k
        for (int y = row_start8; y < row_end8; y++) {
719
163k
            const int y_sb_align = y & ~7;
720
163k
            const int y_proj_start = imax(y_sb_align, row_start8);
721
163k
            const int y_proj_end = imin(y_sb_align + 8, row_end8);
722
343k
            for (int x = col_start8i; x < col_end8i; x++) {
723
180k
                const refmvs_temporal_block *rb = &r[x];
724
180k
                const int b_ref = rb->ref;
725
180k
                if (!b_ref) continue;
726
54.5k
                const int ref2ref = rf->mfmv_ref2ref[n][b_ref - 1];
727
54.5k
                if (!ref2ref) continue;
728
47.3k
                const mv b_mv = rb->mv;
729
47.3k
                const mv offset = mv_projection(b_mv, ref2cur, ref2ref);
730
47.3k
                int pos_x = x + apply_sign(abs(offset.x) >> 6,
731
47.3k
                                           offset.x ^ ref_sign);
732
47.3k
                const int pos_y = y + apply_sign(abs(offset.y) >> 6,
733
47.3k
                                                 offset.y ^ ref_sign);
734
47.3k
                if (pos_y >= y_proj_start && pos_y < y_proj_end) {
735
41.4k
                    const ptrdiff_t pos = (pos_y & 15) * stride;
736
59.5k
                    for (;;) {
737
59.5k
                        const int x_sb_align = x & ~7;
738
59.5k
                        if (pos_x >= imax(x_sb_align - 8, col_start8) &&
739
59.1k
                            pos_x < imin(x_sb_align + 16, col_end8))
740
58.8k
                        {
741
58.8k
                            rp_proj[pos + pos_x].mv = rb->mv;
742
58.8k
                            rp_proj[pos + pos_x].ref = ref2ref;
743
58.8k
                        }
744
59.5k
                        if (++x >= col_end8i) break;
745
27.1k
                        rb++;
746
27.1k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
747
18.0k
                        pos_x++;
748
18.0k
                    }
749
41.4k
                } else {
750
6.36k
                    for (;;) {
751
6.36k
                        if (++x >= col_end8i) break;
752
3.42k
                        rb++;
753
3.42k
                        if (rb->ref != b_ref || rb->mv.n != b_mv.n) break;
754
3.42k
                    }
755
5.87k
                }
756
47.3k
                x--;
757
47.3k
            }
758
163k
            r += stride;
759
163k
        }
760
24.5k
    }
761
20.6k
}
762
763
static void save_tmvs_c(refmvs_temporal_block *rp, const ptrdiff_t stride,
764
                        refmvs_block *const *const rr,
765
                        const uint8_t *const ref_sign,
766
                        const int col_end8, const int row_end8,
767
                        const int col_start8, const int row_start8)
768
42.6k
{
769
247k
    for (int y = row_start8; y < row_end8; y++) {
770
204k
        const refmvs_block *const b = rr[(y & 15) * 2];
771
772
418k
        for (int x = col_start8; x < col_end8;) {
773
214k
            const refmvs_block *const cand_b = &b[x * 2 + 1];
774
214k
            const int bw8 = (dav1d_block_dimensions[cand_b->bs][0] + 1) >> 1;
775
776
214k
            if (cand_b->ref.ref[1] > 0 && ref_sign[cand_b->ref.ref[1] - 1] &&
777
12.8k
                (abs(cand_b->mv.mv[1].y) | abs(cand_b->mv.mv[1].x)) < 4096)
778
12.6k
            {
779
12.6k
                const refmvs_temporal_block tmv = {
780
12.6k
                    .mv = cand_b->mv.mv[1],
781
12.6k
                    .ref = cand_b->ref.ref[1],
782
12.6k
                };
783
58.0k
                for (int n = 0; n < bw8; n++, x++)
784
45.4k
                    rp[x] = tmv;
785
201k
            } else if (cand_b->ref.ref[0] > 0 && ref_sign[cand_b->ref.ref[0] - 1] &&
786
29.7k
                       (abs(cand_b->mv.mv[0].y) | abs(cand_b->mv.mv[0].x)) < 4096)
787
29.4k
            {
788
29.4k
                const refmvs_temporal_block tmv = {
789
29.4k
                    .mv = cand_b->mv.mv[0],
790
29.4k
                    .ref = cand_b->ref.ref[0],
791
29.4k
                };
792
193k
                for (int n = 0; n < bw8; n++, x++)
793
164k
                    rp[x] = tmv;
794
172k
            } else {
795
172k
                const refmvs_temporal_block tmv = { .mv = { .n = 0 }, .ref = 0 };
796
1.10M
                for (int n = 0; n < bw8; n++, x++)
797
928k
                    rp[x] = tmv;
798
172k
            }
799
214k
        }
800
204k
        rp += stride;
801
204k
    }
802
42.6k
}
803
804
int dav1d_refmvs_init_frame(refmvs_frame *const rf,
805
                            const Dav1dSequenceHeader *const seq_hdr,
806
                            const Dav1dFrameHeader *const frm_hdr,
807
                            const uint8_t ref_poc[7],
808
                            refmvs_temporal_block *const rp,
809
                            const uint8_t ref_ref_poc[7][7],
810
                            /*const*/ refmvs_temporal_block *const rp_ref[7],
811
                            const int n_tile_threads, const int n_frame_threads)
812
100k
{
813
100k
    const int rp_stride = ((frm_hdr->width[0] + 127) & ~127) >> 3;
814
100k
    const int n_tile_rows = n_tile_threads > 1 ? frm_hdr->tiling.rows : 1;
815
100k
    const int n_blocks = rp_stride * n_tile_rows;
816
817
100k
    rf->sbsz = 16 << seq_hdr->sb128;
818
100k
    rf->frm_hdr = frm_hdr;
819
100k
    rf->iw8 = (frm_hdr->width[0] + 7) >> 3;
820
100k
    rf->ih8 = (frm_hdr->height + 7) >> 3;
821
100k
    rf->iw4 = rf->iw8 << 1;
822
100k
    rf->ih4 = rf->ih8 << 1;
823
100k
    rf->rp = rp;
824
100k
    rf->rp_stride = rp_stride;
825
100k
    rf->n_tile_threads = n_tile_threads;
826
100k
    rf->n_frame_threads = n_frame_threads;
827
828
100k
    if (n_blocks != rf->n_blocks) {
829
28.4k
        const size_t r_sz = sizeof(*rf->r) * 35 * 2 * n_blocks * (1 + (n_frame_threads > 1));
830
28.4k
        const size_t rp_proj_sz = sizeof(*rf->rp_proj) * 16 * n_blocks;
831
        /* Note that sizeof(*rf->r) == 12, but it's accessed using 16-byte unaligned
832
         * loads in save_tmvs() asm which can overread 4 bytes into rp_proj. */
833
28.4k
        dav1d_free_aligned(rf->r);
834
28.4k
        rf->r = dav1d_alloc_aligned(ALLOC_REFMVS, r_sz + rp_proj_sz, 64);
835
28.4k
        if (!rf->r) {
836
0
            rf->n_blocks = 0;
837
0
            return DAV1D_ERR(ENOMEM);
838
0
        }
839
840
28.4k
        rf->rp_proj = (refmvs_temporal_block*)((uintptr_t)rf->r + r_sz);
841
28.4k
        rf->n_blocks = n_blocks;
842
28.4k
    }
843
844
100k
    const int poc = frm_hdr->frame_offset;
845
804k
    for (int i = 0; i < 7; i++) {
846
704k
        const int poc_diff = get_poc_diff(seq_hdr->order_hint_n_bits,
847
704k
                                          ref_poc[i], poc);
848
704k
        rf->sign_bias[i] = poc_diff > 0;
849
704k
        rf->mfmv_sign[i] = poc_diff < 0;
850
704k
        rf->pocdiff[i] = iclip(get_poc_diff(seq_hdr->order_hint_n_bits,
851
704k
                                            poc, ref_poc[i]), -31, 31);
852
704k
    }
853
854
    // temporal MV setup
855
100k
    rf->n_mfmvs = 0;
856
100k
    rf->rp_ref = rp_ref;
857
100k
    if (frm_hdr->use_ref_frame_mvs && seq_hdr->order_hint_n_bits) {
858
20.5k
        int total = 2;
859
20.5k
        if (rp_ref[0] && ref_ref_poc[0][6] != ref_poc[3] /* alt-of-last != gold */) {
860
6.08k
            rf->mfmv_ref[rf->n_mfmvs++] = 0; // last
861
6.08k
            total = 3;
862
6.08k
        }
863
20.5k
        if (rp_ref[4] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[4],
864
15.7k
                                      frm_hdr->frame_offset) > 0)
865
1.12k
        {
866
1.12k
            rf->mfmv_ref[rf->n_mfmvs++] = 4; // bwd
867
1.12k
        }
868
20.5k
        if (rp_ref[5] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[5],
869
17.4k
                                      frm_hdr->frame_offset) > 0)
870
355
        {
871
355
            rf->mfmv_ref[rf->n_mfmvs++] = 5; // altref2
872
355
        }
873
20.5k
        if (rf->n_mfmvs < total && rp_ref[6] &&
874
8.35k
            get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[6],
875
8.35k
                         frm_hdr->frame_offset) > 0)
876
1.94k
        {
877
1.94k
            rf->mfmv_ref[rf->n_mfmvs++] = 6; // altref
878
1.94k
        }
879
20.5k
        if (rf->n_mfmvs < total && rp_ref[1])
880
16.9k
            rf->mfmv_ref[rf->n_mfmvs++] = 1; // last2
881
882
47.0k
        for (int n = 0; n < rf->n_mfmvs; n++) {
883
26.4k
            const int rpoc = ref_poc[rf->mfmv_ref[n]];
884
26.4k
            const int diff1 = get_poc_diff(seq_hdr->order_hint_n_bits,
885
26.4k
                                           rpoc, frm_hdr->frame_offset);
886
26.4k
            if (abs(diff1) > 31) {
887
1.63k
                rf->mfmv_ref2cur[n] = INVALID_REF2CUR;
888
24.8k
            } else {
889
24.8k
                rf->mfmv_ref2cur[n] = rf->mfmv_ref[n] < 4 ? -diff1 : diff1;
890
198k
                for (int m = 0; m < 7; m++) {
891
174k
                    const int rrpoc = ref_ref_poc[rf->mfmv_ref[n]][m];
892
174k
                    const int diff2 = get_poc_diff(seq_hdr->order_hint_n_bits,
893
174k
                                                   rpoc, rrpoc);
894
                    // unsigned comparison also catches the < 0 case
895
174k
                    rf->mfmv_ref2ref[n][m] = (unsigned) diff2 > 31U ? 0 : diff2;
896
174k
                }
897
24.8k
            }
898
26.4k
        }
899
20.5k
    }
900
100k
    rf->use_ref_frame_mvs = rf->n_mfmvs > 0;
901
902
100k
    return 0;
903
100k
}
904
905
static void splat_mv_c(refmvs_block **rr, const refmvs_block *const rmv,
906
                       const int bx4, const int bw4, int bh4)
907
1.34M
{
908
5.74M
    do {
909
5.74M
        refmvs_block *const r = *rr++ + bx4;
910
57.3M
        for (int x = 0; x < bw4; x++)
911
51.6M
            r[x] = *rmv;
912
5.74M
    } while (--bh4);
913
1.34M
}
914
915
#if HAVE_ASM
916
#if ARCH_AARCH64 || ARCH_ARM
917
#include "src/arm/refmvs.h"
918
#elif ARCH_LOONGARCH64
919
#include "src/loongarch/refmvs.h"
920
#elif ARCH_X86
921
#include "src/x86/refmvs.h"
922
#endif
923
#endif
924
925
COLD void dav1d_refmvs_dsp_init(Dav1dRefmvsDSPContext *const c)
926
24.6k
{
927
24.6k
    c->load_tmvs = load_tmvs_c;
928
24.6k
    c->save_tmvs = save_tmvs_c;
929
24.6k
    c->splat_mv = splat_mv_c;
930
931
#if HAVE_ASM
932
#if ARCH_AARCH64 || ARCH_ARM
933
    refmvs_dsp_init_arm(c);
934
#elif ARCH_LOONGARCH64
935
    refmvs_dsp_init_loongarch(c);
936
#elif ARCH_X86
937
    refmvs_dsp_init_x86(c);
938
#endif
939
#endif
940
24.6k
}