Coverage Report

Created: 2026-09-06 07:31

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/dav1d/src/thread_task.c
Line
Count
Source
1
/*
2
 * Copyright © 2018, VideoLAN and dav1d authors
3
 * Copyright © 2018, Two Orioles, LLC
4
 * All rights reserved.
5
 *
6
 * Redistribution and use in source and binary forms, with or without
7
 * modification, are permitted provided that the following conditions are met:
8
 *
9
 * 1. Redistributions of source code must retain the above copyright notice, this
10
 *    list of conditions and the following disclaimer.
11
 *
12
 * 2. Redistributions in binary form must reproduce the above copyright notice,
13
 *    this list of conditions and the following disclaimer in the documentation
14
 *    and/or other materials provided with the distribution.
15
 *
16
 * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
17
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
18
 * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
19
 * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
20
 * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
21
 * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
22
 * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
23
 * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
24
 * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
25
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
26
 */
27
28
#include "config.h"
29
30
#include "common/frame.h"
31
32
#include "src/thread_task.h"
33
#include "src/fg_apply.h"
34
35
// This function resets the cur pointer to the first frame theoretically
36
// executable after a task completed (ie. each time we update some progress or
37
// insert some tasks in the queue).
38
// When frame_idx is set, it can be either from a completed task, or from tasks
39
// inserted in the queue, in which case we have to make sure the cur pointer
40
// isn't past this insert.
41
// The special case where frame_idx is UINT_MAX is to handle the reset after
42
// completing a task and locklessly signaling progress. In this case we don't
43
// enter a critical section, which is needed for this function, so we set an
44
// atomic for a delayed handling, happening here. Meaning we can call this
45
// function without any actual update other than what's in the atomic, hence
46
// this special case.
47
static inline int reset_task_cur(const Dav1dContext *const c,
48
                                 struct TaskThreadData *const ttd,
49
                                 unsigned frame_idx)
50
7.35M
{
51
7.35M
    const unsigned first = atomic_load(&ttd->first);
52
7.35M
    unsigned reset_frame_idx = atomic_exchange(&ttd->reset_task_cur, UINT_MAX);
53
7.35M
    if (reset_frame_idx < first) {
54
0
        if (frame_idx == UINT_MAX) return 0;
55
0
        reset_frame_idx = UINT_MAX;
56
0
    }
57
7.35M
    if (!ttd->cur && c->fc[first].task_thread.task_cur_prev == NULL)
58
1.99M
        return 0;
59
5.35M
    if (reset_frame_idx != UINT_MAX) {
60
141k
        if (frame_idx == UINT_MAX) {
61
74.3k
            if (reset_frame_idx > first + ttd->cur)
62
208
                return 0;
63
74.0k
            ttd->cur = reset_frame_idx - first;
64
74.0k
            goto cur_found;
65
74.3k
        }
66
5.21M
    } else if (frame_idx == UINT_MAX)
67
4.23M
        return 0;
68
1.04M
    if (frame_idx < first) frame_idx += c->n_fc;
69
1.04M
    const unsigned min_frame_idx = umin(reset_frame_idx, frame_idx);
70
1.04M
    const unsigned cur_frame_idx = first + ttd->cur;
71
1.04M
    if (ttd->cur < c->n_fc && cur_frame_idx < min_frame_idx)
72
43.7k
        return 0;
73
1.61M
    for (ttd->cur = min_frame_idx - first; ttd->cur < c->n_fc; ttd->cur++)
74
1.52M
        if (c->fc[(first + ttd->cur) % c->n_fc].task_thread.task_head)
75
908k
            break;
76
1.08M
cur_found:
77
6.53M
    for (unsigned i = ttd->cur; i < c->n_fc; i++)
78
5.45M
        c->fc[(first + i) % c->n_fc].task_thread.task_cur_prev = NULL;
79
1.08M
    return 1;
80
1.00M
}
81
82
static inline void reset_task_cur_async(struct TaskThreadData *const ttd,
83
                                        unsigned frame_idx, unsigned n_frames)
84
702k
{
85
702k
    const unsigned first = atomic_load(&ttd->first);
86
702k
    if (frame_idx < first) frame_idx += n_frames;
87
702k
    unsigned last_idx = frame_idx;
88
702k
    do {
89
702k
        frame_idx = last_idx;
90
702k
        last_idx = atomic_exchange(&ttd->reset_task_cur, frame_idx);
91
702k
    } while (last_idx < frame_idx);
92
702k
    if (frame_idx == first && atomic_load(&ttd->first) != first) {
93
0
        unsigned expected = frame_idx;
94
0
        atomic_compare_exchange_strong(&ttd->reset_task_cur, &expected, UINT_MAX);
95
0
    }
96
702k
}
97
98
static void insert_tasks_between(Dav1dFrameContext *const f,
99
                                 Dav1dTask *const first, Dav1dTask *const last,
100
                                 Dav1dTask *const a, Dav1dTask *const b,
101
                                 const int cond_signal)
102
1.52M
{
103
1.52M
    struct TaskThreadData *const ttd = f->task_thread.ttd;
104
1.52M
    if (atomic_load(f->c->flush)) return;
105
1.52M
    assert(!a || a->next == b);
106
1.52M
    if (!a) f->task_thread.task_head = first;
107
1.12M
    else a->next = first;
108
1.52M
    if (!b) f->task_thread.task_tail = last;
109
1.52M
    last->next = b;
110
1.52M
    reset_task_cur(f->c, ttd, first->frame_idx);
111
1.52M
    if (cond_signal && !atomic_fetch_or(&ttd->cond_signaled, 1))
112
44.7k
        pthread_cond_signal(&ttd->cond);
113
1.52M
}
114
115
static void insert_tasks(Dav1dFrameContext *const f,
116
                         Dav1dTask *const first, Dav1dTask *const last,
117
                         const int cond_signal)
118
1.52M
{
119
    // insert task back into task queue
120
1.52M
    Dav1dTask *t_ptr, *prev_t = NULL;
121
1.52M
    for (t_ptr = f->task_thread.task_head;
122
5.80M
         t_ptr; prev_t = t_ptr, t_ptr = t_ptr->next)
123
4.80M
    {
124
        // entropy coding precedes other steps
125
4.80M
        if (t_ptr->type == DAV1D_TASK_TYPE_TILE_ENTROPY) {
126
980k
            if (first->type > DAV1D_TASK_TYPE_TILE_ENTROPY) continue;
127
            // both are entropy
128
259k
            if (first->sby > t_ptr->sby) continue;
129
38.6k
            if (first->sby < t_ptr->sby) {
130
941
                insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
131
941
                return;
132
941
            }
133
            // same sby
134
3.82M
        } else {
135
3.82M
            if (first->type == DAV1D_TASK_TYPE_TILE_ENTROPY) {
136
246k
                insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
137
246k
                return;
138
246k
            }
139
3.57M
            if (first->sby > t_ptr->sby) continue;
140
834k
            if (first->sby < t_ptr->sby) {
141
272k
                insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
142
272k
                return;
143
272k
            }
144
            // same sby
145
562k
            if (first->type > t_ptr->type) continue;
146
42.3k
            if (first->type < t_ptr->type) {
147
4.44k
                insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
148
4.44k
                return;
149
4.44k
            }
150
            // same task type
151
42.3k
        }
152
153
        // sort by tile-id
154
75.6k
        assert(first->type == DAV1D_TASK_TYPE_TILE_RECONSTRUCTION ||
155
75.6k
               first->type == DAV1D_TASK_TYPE_TILE_ENTROPY);
156
75.6k
        assert(first->type == t_ptr->type);
157
75.6k
        assert(t_ptr->sby == first->sby);
158
75.6k
        const int p = first->type == DAV1D_TASK_TYPE_TILE_ENTROPY;
159
75.6k
        const int t_tile_idx = (int) (first - f->task_thread.tile_tasks[p]);
160
75.6k
        const int p_tile_idx = (int) (t_ptr - f->task_thread.tile_tasks[p]);
161
75.6k
        assert(t_tile_idx != p_tile_idx);
162
75.6k
        if (t_tile_idx > p_tile_idx) continue;
163
133
        insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
164
133
        return;
165
75.6k
    }
166
    // append at the end
167
997k
    insert_tasks_between(f, first, last, prev_t, NULL, cond_signal);
168
997k
}
169
170
static inline void insert_task(Dav1dFrameContext *const f,
171
                               Dav1dTask *const t, const int cond_signal)
172
1.52M
{
173
1.52M
    insert_tasks(f, t, t, cond_signal);
174
1.52M
}
175
176
487k
static inline void add_pending(Dav1dFrameContext *const f, Dav1dTask *const t) {
177
487k
    pthread_mutex_lock(&f->task_thread.pending_tasks.lock);
178
487k
    t->next = NULL;
179
487k
    if (!f->task_thread.pending_tasks.head)
180
470k
        f->task_thread.pending_tasks.head = t;
181
17.2k
    else
182
17.2k
        f->task_thread.pending_tasks.tail->next = t;
183
487k
    f->task_thread.pending_tasks.tail = t;
184
487k
    atomic_store(&f->task_thread.pending_tasks.merge, 1);
185
487k
    pthread_mutex_unlock(&f->task_thread.pending_tasks.lock);
186
487k
}
187
188
43.0M
static inline int merge_pending_frame(Dav1dFrameContext *const f) {
189
43.0M
    int const merge = atomic_load(&f->task_thread.pending_tasks.merge);
190
43.0M
    if (merge) {
191
533k
        pthread_mutex_lock(&f->task_thread.pending_tasks.lock);
192
533k
        Dav1dTask *t = f->task_thread.pending_tasks.head;
193
533k
        f->task_thread.pending_tasks.head = NULL;
194
533k
        f->task_thread.pending_tasks.tail = NULL;
195
533k
        atomic_store(&f->task_thread.pending_tasks.merge, 0);
196
533k
        pthread_mutex_unlock(&f->task_thread.pending_tasks.lock);
197
1.28M
        while (t) {
198
753k
            Dav1dTask *const tmp = t->next;
199
753k
            insert_task(f, t, 0);
200
753k
            t = tmp;
201
753k
        }
202
533k
    }
203
43.0M
    return merge;
204
43.0M
}
205
206
6.37M
static inline int merge_pending(const Dav1dContext *const c) {
207
6.37M
    int res = 0;
208
44.6M
    for (unsigned i = 0; i < c->n_fc; i++)
209
38.2M
        res |= merge_pending_frame(&c->fc[i]);
210
6.37M
    return res;
211
6.37M
}
212
213
static int create_filter_sbrow(Dav1dFrameContext *const f,
214
                               const int pass, Dav1dTask **res_t)
215
126k
{
216
126k
    const int has_deblock = f->frame_hdr->loopfilter.level_y[0] ||
217
44.2k
                            f->frame_hdr->loopfilter.level_y[1];
218
126k
    const int has_cdef = f->seq_hdr->cdef;
219
126k
    const int has_resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
220
126k
    const int has_lr = f->lf.restore_planes;
221
222
126k
    Dav1dTask *tasks = f->task_thread.tasks;
223
126k
    const int uses_2pass = f->c->n_fc > 1;
224
126k
    int num_tasks = f->sbh * (1 + uses_2pass);
225
126k
    if (num_tasks > f->task_thread.num_tasks) {
226
53.2k
        const size_t size = sizeof(Dav1dTask) * num_tasks;
227
53.2k
        tasks = dav1d_realloc(ALLOC_COMMON_CTX, f->task_thread.tasks, size);
228
53.2k
        if (!tasks) return -1;
229
53.2k
        memset(tasks, 0, size);
230
53.2k
        f->task_thread.tasks = tasks;
231
53.2k
        f->task_thread.num_tasks = num_tasks;
232
53.2k
    }
233
126k
    tasks += f->sbh * (pass & 1);
234
235
126k
    if (pass & 1) {
236
63.3k
        f->frame_thread.entropy_progress = 0;
237
63.3k
    } else {
238
63.2k
        const int prog_sz = ((f->sbh + 31) & ~31) >> 5;
239
63.2k
        if (prog_sz > f->frame_thread.prog_sz) {
240
53.0k
            atomic_uint *const prog = dav1d_realloc(ALLOC_COMMON_CTX, f->frame_thread.frame_progress,
241
53.0k
                                                    2 * prog_sz * sizeof(*prog));
242
53.0k
            if (!prog) return -1;
243
53.0k
            f->frame_thread.frame_progress = prog;
244
53.0k
            f->frame_thread.copy_lpf_progress = prog + prog_sz;
245
53.0k
        }
246
63.2k
        f->frame_thread.prog_sz = prog_sz;
247
63.2k
        memset(f->frame_thread.frame_progress, 0, prog_sz * sizeof(atomic_uint));
248
63.2k
        memset(f->frame_thread.copy_lpf_progress, 0, prog_sz * sizeof(atomic_uint));
249
63.2k
        atomic_store(&f->frame_thread.deblock_progress, 0);
250
63.2k
    }
251
126k
    f->frame_thread.next_tile_row[pass & 1] = 0;
252
253
126k
    Dav1dTask *t = &tasks[0];
254
126k
    t->sby = 0;
255
126k
    t->recon_progress = 1;
256
126k
    t->deblock_progress = 0;
257
126k
    t->type = pass == 1 ? DAV1D_TASK_TYPE_ENTROPY_PROGRESS :
258
126k
              has_deblock ? DAV1D_TASK_TYPE_DEBLOCK_COLS :
259
63.2k
              has_cdef || has_lr /* i.e. LR backup */ ? DAV1D_TASK_TYPE_DEBLOCK_ROWS :
260
18.5k
              has_resize ? DAV1D_TASK_TYPE_SUPER_RESOLUTION :
261
9.46k
              DAV1D_TASK_TYPE_RECONSTRUCTION_PROGRESS;
262
126k
    t->frame_idx = (int)(f - f->c->fc);
263
264
126k
    *res_t = t;
265
126k
    return 0;
266
126k
}
267
268
int dav1d_task_create_tile_sbrow(Dav1dFrameContext *const f, const int pass,
269
                                 const int cond_signal)
270
126k
{
271
126k
    Dav1dTask *tasks = f->task_thread.tile_tasks[0];
272
126k
    const int uses_2pass = f->c->n_fc > 1;
273
126k
    const int n_tasks_per_pass = f->frame_hdr->tiling.cols * f->frame_hdr->tiling.rows;
274
126k
    const int n_tasks = n_tasks_per_pass * (1 + uses_2pass);
275
126k
    if (pass < 2) {
276
63.2k
        if (n_tasks > f->task_thread.num_tile_tasks) {
277
53.0k
            const size_t size = sizeof(Dav1dTask) * n_tasks;
278
53.0k
            tasks = dav1d_realloc(ALLOC_COMMON_CTX, f->task_thread.tile_tasks[0], size);
279
53.0k
            if (!tasks) return -1;
280
53.0k
            memset(tasks, 0, size);
281
53.0k
            f->task_thread.tile_tasks[0] = tasks;
282
53.0k
            f->task_thread.num_tile_tasks = n_tasks;
283
53.0k
        }
284
63.2k
        f->task_thread.tile_tasks[1] = tasks + n_tasks_per_pass;
285
63.2k
    }
286
126k
    assert(n_tasks <= f->task_thread.num_tile_tasks);
287
288
126k
    Dav1dTask *pf_t;
289
126k
    if (create_filter_sbrow(f, pass, &pf_t))
290
0
        return -1;
291
292
126k
    Dav1dTask *const p1_tasks = f->task_thread.tile_tasks[1];
293
126k
    Dav1dTask *prev_t = NULL;
294
126k
    if (pass == 2) {
295
63.3k
        prev_t = &p1_tasks[n_tasks_per_pass - 1];
296
        // PF task is scheduled after the last sby=0 TILE task
297
63.3k
        if (f->frame_hdr->tiling.rows == 1)
298
62.6k
            prev_t = prev_t->next;
299
63.3k
    }
300
126k
    tasks += (pass & 1) * n_tasks_per_pass;
301
266k
    for (int tile_idx = 0; tile_idx < n_tasks_per_pass; tile_idx++) {
302
140k
        Dav1dTileState *const ts = &f->ts[tile_idx];
303
140k
        Dav1dTask *t = &tasks[tile_idx];
304
140k
        t->sby = ts->tiling.row_start >> f->sb_shift;
305
140k
        if (pf_t && t->sby) {
306
1.39k
            prev_t->next = pf_t;
307
1.39k
            prev_t = pf_t;
308
1.39k
            pf_t = NULL;
309
1.39k
        }
310
140k
        t->recon_progress = 0;
311
140k
        t->deblock_progress = 0;
312
140k
        t->deps_skip = 0;
313
140k
        t->type = pass != 1 ? DAV1D_TASK_TYPE_TILE_RECONSTRUCTION :
314
140k
                              DAV1D_TASK_TYPE_TILE_ENTROPY;
315
140k
        t->frame_idx = (int)(f - f->c->fc);
316
140k
        if (prev_t) prev_t->next = t;
317
140k
        prev_t = t;
318
140k
    }
319
126k
    if (pf_t) {
320
125k
        prev_t->next = pf_t;
321
125k
        prev_t = pf_t;
322
125k
    }
323
126k
    prev_t->next = NULL;
324
325
126k
    atomic_store(&f->task_thread.done[pass & 1], 0);
326
327
    // XXX in theory this could be done locklessly, at this point they are no
328
    // tasks in the frameQ, so no other runner should be using this lock, but
329
    // we must add both passes at once
330
126k
    if (!(pass & 1)) {
331
63.3k
        pthread_mutex_lock(&f->task_thread.pending_tasks.lock);
332
63.3k
        assert(f->task_thread.pending_tasks.head == NULL);
333
63.3k
        f->task_thread.pending_tasks.head = f->task_thread.tile_tasks[pass == 2];
334
63.3k
        f->task_thread.pending_tasks.tail = prev_t;
335
63.3k
        atomic_store(&f->task_thread.pending_tasks.merge, 1);
336
63.3k
        atomic_store(&f->task_thread.init_done, 1);
337
63.3k
        pthread_mutex_unlock(&f->task_thread.pending_tasks.lock);
338
63.3k
    }
339
126k
    return 0;
340
126k
}
341
342
65.6k
void dav1d_task_frame_init(Dav1dFrameContext *const f) {
343
65.6k
    const Dav1dContext *const c = f->c;
344
345
65.6k
    atomic_store(&f->task_thread.init_done, 0);
346
    // schedule init task, which will schedule the remaining tasks
347
65.6k
    Dav1dTask *const t = &f->task_thread.init_task;
348
65.6k
    t->type = DAV1D_TASK_TYPE_INIT;
349
65.6k
    t->frame_idx = (int)(f - c->fc);
350
65.6k
    t->sby = 0;
351
65.6k
    t->recon_progress = t->deblock_progress = 0;
352
65.6k
    insert_task(f, t, 1);
353
65.6k
}
354
355
void dav1d_task_delayed_fg(Dav1dContext *const c, Dav1dPicture *const out,
356
                           const Dav1dPicture *const in)
357
2.85k
{
358
2.85k
    struct TaskThreadData *const ttd = &c->task_thread;
359
2.85k
    ttd->delayed_fg.in = in;
360
2.85k
    ttd->delayed_fg.out = out;
361
2.85k
    ttd->delayed_fg.type = DAV1D_TASK_TYPE_FG_PREP;
362
2.85k
    atomic_init(&ttd->delayed_fg.progress[0], 0);
363
2.85k
    atomic_init(&ttd->delayed_fg.progress[1], 0);
364
2.85k
    pthread_mutex_lock(&ttd->lock);
365
2.85k
    ttd->delayed_fg.exec = 1;
366
2.85k
    ttd->delayed_fg.finished = 0;
367
2.85k
    pthread_cond_signal(&ttd->cond);
368
2.85k
    do {
369
2.85k
        pthread_cond_wait(&ttd->delayed_fg.cond, &ttd->lock);
370
2.85k
    } while (!ttd->delayed_fg.finished);
371
2.85k
    pthread_mutex_unlock(&ttd->lock);
372
2.85k
}
373
374
static inline int ensure_progress(struct TaskThreadData *const ttd,
375
                                  Dav1dFrameContext *const f,
376
                                  Dav1dTask *const t, const enum TaskType type,
377
                                  atomic_int *const state, int *const target)
378
255k
{
379
    // deblock_rows (non-LR portion) depends on deblock of previous sbrow,
380
    // so ensure that completed. if not, re-add to task-queue; else, fall-through
381
255k
    int p1 = atomic_load(state);
382
255k
    if (p1 < t->sby) {
383
15.3k
        t->type = type;
384
15.3k
        t->recon_progress = t->deblock_progress = 0;
385
15.3k
        *target = t->sby;
386
15.3k
        add_pending(f, t);
387
15.3k
        pthread_mutex_lock(&ttd->lock);
388
15.3k
        return 1;
389
15.3k
    }
390
240k
    return 0;
391
255k
}
392
393
static inline int check_tile(Dav1dTask *const t, Dav1dFrameContext *const f,
394
                             const int frame_mt)
395
1.51M
{
396
1.51M
    const int tp = t->type == DAV1D_TASK_TYPE_TILE_ENTROPY;
397
1.51M
    const int tile_idx = (int)(t - f->task_thread.tile_tasks[tp]);
398
1.51M
    Dav1dTileState *const ts = &f->ts[tile_idx];
399
1.51M
    const int p1 = atomic_load(&ts->progress[tp]);
400
1.51M
    if (p1 < t->sby) return 1;
401
1.05M
    int error = p1 == TILE_ERROR;
402
1.05M
    error |= atomic_fetch_or(&f->task_thread.error, error);
403
1.05M
    if (!error && frame_mt && !tp) {
404
438k
        const int p2 = atomic_load(&ts->progress[1]);
405
438k
        if (p2 <= t->sby) return 1;
406
307k
        error = p2 == TILE_ERROR;
407
307k
        error |= atomic_fetch_or(&f->task_thread.error, error);
408
307k
    }
409
926k
    if (!error && frame_mt && !IS_KEY_OR_INTRA(f->frame_hdr)) {
410
        // check reference state
411
145k
        const Dav1dThreadPicture *p = &f->sr_cur;
412
145k
        const int ss_ver = p->p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
413
145k
        const unsigned p_b = (t->sby + 1) << (f->sb_shift + 2);
414
145k
        const int tile_sby = t->sby - (ts->tiling.row_start >> f->sb_shift);
415
145k
        const int (*const lowest_px)[2] = ts->lowest_pixel[tile_sby];
416
558k
        for (int n = t->deps_skip; n < 7; n++, t->deps_skip++) {
417
500k
            unsigned lowest;
418
500k
            if (tp) {
419
                // if temporal mv refs are disabled, we only need this
420
                // for the primary ref; if segmentation is disabled, we
421
                // don't even need that
422
255k
                lowest = p_b;
423
255k
            } else {
424
                // +8 is postfilter-induced delay
425
244k
                const int y = lowest_px[n][0] == INT_MIN ? INT_MIN :
426
244k
                              lowest_px[n][0] + 8;
427
244k
                const int uv = lowest_px[n][1] == INT_MIN ? INT_MIN :
428
244k
                               lowest_px[n][1] * (1 << ss_ver) + 8;
429
244k
                const int max = imax(y, uv);
430
244k
                if (max == INT_MIN) continue;
431
95.0k
                lowest = iclip(max, 1, f->refp[n].p.p.h);
432
95.0k
            }
433
350k
            const unsigned p3 = atomic_load(&f->refp[n].progress[!tp]);
434
350k
            if (p3 < lowest) return 1;
435
350k
            atomic_fetch_or(&f->task_thread.error, p3 == FRAME_ERROR);
436
263k
        }
437
145k
    }
438
839k
    return 0;
439
926k
}
440
441
static inline int get_frame_progress(const Dav1dContext *const c,
442
                                     const Dav1dFrameContext *const f)
443
412k
{
444
412k
    unsigned frame_prog = c->n_fc > 1 ? atomic_load(&f->sr_cur.progress[1]) : 0;
445
412k
    if (frame_prog >= FRAME_ERROR)
446
150k
        return f->sbh - 1;
447
261k
    int idx = frame_prog >> (f->sb_shift + 7);
448
261k
    int prog;
449
265k
    do {
450
265k
        atomic_uint *state = &f->frame_thread.frame_progress[idx];
451
265k
        const unsigned val = ~atomic_load(state);
452
265k
        prog = val ? ctz(val) : 32;
453
265k
        if (prog != 32) break;
454
4.31k
        prog = 0;
455
4.31k
    } while (++idx < f->frame_thread.prog_sz);
456
261k
    return ((idx << 5) | prog) - 1;
457
412k
}
458
459
1.14k
static inline void abort_frame(Dav1dFrameContext *const f, const int error) {
460
1.14k
    atomic_store(&f->task_thread.error, error == DAV1D_ERR(EINVAL) ? 1 : -1);
461
1.14k
    atomic_store(&f->task_thread.task_counter, 0);
462
1.14k
    atomic_store(&f->task_thread.done[0], 1);
463
1.14k
    atomic_store(&f->task_thread.done[1], 1);
464
1.14k
    atomic_store(&f->sr_cur.progress[0], FRAME_ERROR);
465
1.14k
    atomic_store(&f->sr_cur.progress[1], FRAME_ERROR);
466
1.14k
    dav1d_decode_frame_exit(f, error);
467
1.14k
    f->n_tile_data = 0;
468
1.14k
    pthread_cond_signal(&f->task_thread.cond);
469
1.14k
}
470
471
static inline void delayed_fg_task(const Dav1dContext *const c,
472
                                   struct TaskThreadData *const ttd)
473
18.2k
{
474
18.2k
    const Dav1dPicture *const in = ttd->delayed_fg.in;
475
18.2k
    Dav1dPicture *const out = ttd->delayed_fg.out;
476
18.2k
#if CONFIG_16BPC
477
18.2k
    int off;
478
18.2k
    if (out->p.bpc != 8)
479
11.8k
        off = (out->p.bpc >> 1) - 4;
480
18.2k
#endif
481
18.2k
    switch (ttd->delayed_fg.type) {
482
2.85k
    case DAV1D_TASK_TYPE_FG_PREP:
483
2.85k
        ttd->delayed_fg.exec = 0;
484
2.85k
        if (atomic_load(&ttd->cond_signaled))
485
2
            pthread_cond_signal(&ttd->cond);
486
2.85k
        pthread_mutex_unlock(&ttd->lock);
487
2.85k
        switch (out->p.bpc) {
488
0
#if CONFIG_8BPC
489
1.21k
        case 8:
490
1.21k
            dav1d_prep_grain_8bpc(&c->dsp[0].fg, out, in,
491
1.21k
                                  ttd->delayed_fg.scaling_8bpc,
492
1.21k
                                  ttd->delayed_fg.grain_lut_8bpc);
493
1.21k
            break;
494
0
#endif
495
0
#if CONFIG_16BPC
496
1.58k
        case 10:
497
1.64k
        case 12:
498
1.64k
            dav1d_prep_grain_16bpc(&c->dsp[off].fg, out, in,
499
1.64k
                                   ttd->delayed_fg.scaling_16bpc,
500
1.64k
                                   ttd->delayed_fg.grain_lut_16bpc);
501
1.64k
            break;
502
0
#endif
503
0
        default: abort();
504
2.85k
        }
505
2.85k
        ttd->delayed_fg.type = DAV1D_TASK_TYPE_FG_APPLY;
506
2.85k
        pthread_mutex_lock(&ttd->lock);
507
2.85k
        ttd->delayed_fg.exec = 1;
508
        // fall-through
509
18.2k
    case DAV1D_TASK_TYPE_FG_APPLY:;
510
18.2k
        int row = atomic_fetch_add(&ttd->delayed_fg.progress[0], 1);
511
18.2k
        pthread_mutex_unlock(&ttd->lock);
512
18.2k
        int progmax = (out->p.h + FG_BLOCK_SIZE - 1) / FG_BLOCK_SIZE;
513
52.3k
        while (row < progmax) {
514
37.5k
            if (row + 1 < progmax)
515
34.7k
                pthread_cond_signal(&ttd->cond);
516
2.84k
            else {
517
2.84k
                pthread_mutex_lock(&ttd->lock);
518
2.84k
                ttd->delayed_fg.exec = 0;
519
2.84k
                pthread_mutex_unlock(&ttd->lock);
520
2.84k
            }
521
37.5k
            switch (out->p.bpc) {
522
0
#if CONFIG_8BPC
523
11.5k
            case 8:
524
11.5k
                dav1d_apply_grain_row_8bpc(&c->dsp[0].fg, out, in,
525
11.5k
                                           ttd->delayed_fg.scaling_8bpc,
526
11.5k
                                           ttd->delayed_fg.grain_lut_8bpc, row);
527
11.5k
                break;
528
0
#endif
529
0
#if CONFIG_16BPC
530
25.2k
            case 10:
531
26.0k
            case 12:
532
26.0k
                dav1d_apply_grain_row_16bpc(&c->dsp[off].fg, out, in,
533
26.0k
                                            ttd->delayed_fg.scaling_16bpc,
534
26.0k
                                            ttd->delayed_fg.grain_lut_16bpc, row);
535
26.0k
                break;
536
0
#endif
537
0
            default: abort();
538
37.5k
            }
539
34.1k
            row = atomic_fetch_add(&ttd->delayed_fg.progress[0], 1);
540
34.1k
            atomic_fetch_add(&ttd->delayed_fg.progress[1], 1);
541
34.1k
        }
542
14.7k
        pthread_mutex_lock(&ttd->lock);
543
14.7k
        ttd->delayed_fg.exec = 0;
544
14.7k
        int done = atomic_fetch_add(&ttd->delayed_fg.progress[1], 1) + 1;
545
14.7k
        progmax = atomic_load(&ttd->delayed_fg.progress[0]);
546
        // signal for completion only once the last runner reaches this
547
14.7k
        if (done >= progmax) {
548
2.85k
            ttd->delayed_fg.finished = 1;
549
2.85k
            pthread_cond_signal(&ttd->delayed_fg.cond);
550
2.85k
        }
551
14.7k
        break;
552
0
    default: abort();
553
18.2k
    }
554
18.2k
}
555
556
1.18M
void *dav1d_worker_task(void *data) {
557
1.18M
    Dav1dTaskContext *const tc = data;
558
1.18M
    const Dav1dContext *const c = tc->c;
559
1.18M
    struct TaskThreadData *const ttd = tc->task_thread.ttd;
560
561
1.18M
    dav1d_set_thread_name("dav1d-worker");
562
563
1.18M
    pthread_mutex_lock(&ttd->lock);
564
5.17M
    for (;;) {
565
5.17M
        if (tc->task_thread.die) break;
566
3.98M
        if (atomic_load(c->flush)) goto park;
567
568
3.96M
        merge_pending(c);
569
3.96M
        if (ttd->delayed_fg.exec) { // run delayed film grain first
570
18.2k
            delayed_fg_task(c, ttd);
571
18.2k
            continue;
572
18.2k
        }
573
3.94M
        Dav1dFrameContext *f;
574
3.94M
        Dav1dTask *t, *prev_t = NULL;
575
3.94M
        if (c->n_fc > 1) { // run init tasks second
576
27.2M
            for (unsigned i = 0; i < c->n_fc; i++) {
577
23.3M
                const unsigned first = atomic_load(&ttd->first);
578
23.3M
                f = &c->fc[(first + i) % c->n_fc];
579
23.3M
                if (atomic_load(&f->task_thread.init_done)) continue;
580
19.2M
                t = f->task_thread.task_head;
581
19.2M
                if (!t) continue;
582
239k
                if (t->type == DAV1D_TASK_TYPE_INIT) goto found;
583
175k
                if (t->type == DAV1D_TASK_TYPE_INIT_CDF) {
584
                    // XXX This can be a simple else, if adding tasks of both
585
                    // passes at once (in dav1d_task_create_tile_sbrow).
586
                    // Adding the tasks to the pending Q can result in a
587
                    // thread merging them before setting init_done.
588
                    // We will need to set init_done before adding to the
589
                    // pending Q, so maybe return the tasks, set init_done,
590
                    // and add to pending Q only then.
591
175k
                    const int p1 = f->in_cdf.progress ?
592
175k
                        atomic_load(f->in_cdf.progress) : 1;
593
175k
                    if (p1) {
594
11.9k
                        atomic_fetch_or(&f->task_thread.error, p1 == TILE_ERROR);
595
11.9k
                        goto found;
596
11.9k
                    }
597
175k
                }
598
175k
            }
599
3.94M
        }
600
7.19M
        while (ttd->cur < c->n_fc) { // run decoding tasks last
601
4.77M
            const unsigned first = atomic_load(&ttd->first);
602
4.77M
            f = &c->fc[(first + ttd->cur) % c->n_fc];
603
4.77M
            merge_pending_frame(f);
604
4.77M
            prev_t = f->task_thread.task_cur_prev;
605
4.77M
            t = prev_t ? prev_t->next : f->task_thread.task_head;
606
7.77M
            while (t) {
607
4.44M
                if (t->type == DAV1D_TASK_TYPE_INIT_CDF) goto next;
608
4.40M
                else if (t->type == DAV1D_TASK_TYPE_TILE_ENTROPY ||
609
4.05M
                         t->type == DAV1D_TASK_TYPE_TILE_RECONSTRUCTION)
610
809k
                {
611
                    // if not bottom sbrow of tile, this task will be re-added
612
                    // after it's finished
613
809k
                    if (!check_tile(t, f, c->n_fc > 1))
614
592k
                        goto found;
615
3.59M
                } else if (t->recon_progress) {
616
2.01M
                    const int p = t->type == DAV1D_TASK_TYPE_ENTROPY_PROGRESS;
617
2.01M
                    int error = atomic_load(&f->task_thread.error);
618
2.01M
                    assert(!atomic_load(&f->task_thread.done[p]) || error);
619
2.01M
                    const int tile_row_base = f->frame_hdr->tiling.cols *
620
2.01M
                                              f->frame_thread.next_tile_row[p];
621
2.01M
                    if (p) {
622
766k
                        atomic_int *const prog = &f->frame_thread.entropy_progress;
623
766k
                        const int p1 = atomic_load(prog);
624
766k
                        if (p1 < t->sby) goto next;
625
766k
                        atomic_fetch_or(&f->task_thread.error, p1 == TILE_ERROR);
626
671k
                    }
627
2.79M
                    for (int tc = 0; tc < f->frame_hdr->tiling.cols; tc++) {
628
1.96M
                        Dav1dTileState *const ts = &f->ts[tile_row_base + tc];
629
1.96M
                        const int p2 = atomic_load(&ts->progress[p]);
630
1.96M
                        if (p2 < t->recon_progress) goto next;
631
1.96M
                        atomic_fetch_or(&f->task_thread.error, p2 == TILE_ERROR);
632
865k
                    }
633
825k
                    if (t->sby + 1 < f->sbh) {
634
                        // add sby+1 to list to replace this one
635
702k
                        Dav1dTask *next_t = &t[1];
636
702k
                        *next_t = *t;
637
702k
                        next_t->sby++;
638
702k
                        const int ntr = f->frame_thread.next_tile_row[p] + 1;
639
702k
                        const int start = f->frame_hdr->tiling.row_start_sb[ntr];
640
702k
                        if (next_t->sby == start)
641
4.29k
                            f->frame_thread.next_tile_row[p] = ntr;
642
702k
                        next_t->recon_progress = next_t->sby + 1;
643
702k
                        insert_task(f, next_t, 0);
644
702k
                    }
645
825k
                    goto found;
646
1.92M
                } else if (t->type == DAV1D_TASK_TYPE_CDEF) {
647
33.7k
                    atomic_uint *prog = f->frame_thread.copy_lpf_progress;
648
33.7k
                    const int p1 = atomic_load(&prog[(t->sby - 1) >> 5]);
649
33.7k
                    if (p1 & (1U << ((t->sby - 1) & 31)))
650
5.68k
                        goto found;
651
1.54M
                } else {
652
1.54M
                    assert(t->deblock_progress);
653
1.54M
                    const int p1 = atomic_load(&f->frame_thread.deblock_progress);
654
1.54M
                    if (p1 >= t->deblock_progress) {
655
15.3k
                        atomic_fetch_or(&f->task_thread.error, p1 == TILE_ERROR);
656
15.3k
                        goto found;
657
15.3k
                    }
658
1.54M
                }
659
3.00M
            next:
660
3.00M
                prev_t = t;
661
3.00M
                t = t->next;
662
3.00M
                f->task_thread.task_cur_prev = prev_t;
663
3.00M
            }
664
3.33M
            ttd->cur++;
665
3.33M
        }
666
2.42M
        if (reset_task_cur(c, ttd, UINT_MAX)) continue;
667
2.41M
        if (merge_pending(c)) continue;
668
2.43M
    park:
669
2.43M
        tc->task_thread.flushed = 1;
670
2.43M
        pthread_cond_signal(&tc->task_thread.td.cond);
671
        // we want to be woken up next time progress is signaled
672
2.43M
        atomic_store(&ttd->cond_signaled, 0);
673
2.43M
        pthread_cond_wait(&ttd->cond, &ttd->lock);
674
2.43M
        tc->task_thread.flushed = 0;
675
2.43M
        reset_task_cur(c, ttd, UINT_MAX);
676
2.43M
        continue;
677
678
1.51M
    found:
679
        // remove t from list
680
1.51M
        if (prev_t) prev_t->next = t->next;
681
1.22M
        else f->task_thread.task_head = t->next;
682
1.51M
        if (!t->next) f->task_thread.task_tail = prev_t;
683
1.51M
        if (t->type > DAV1D_TASK_TYPE_INIT_CDF && !f->task_thread.task_head)
684
61.8k
            ttd->cur++;
685
1.51M
        t->next = NULL;
686
        // we don't need to check cond_signaled here, since we found a task
687
        // after the last signal so we want to re-signal the next waiting thread
688
        // and again won't need to signal after that
689
1.51M
        atomic_store(&ttd->cond_signaled, 1);
690
1.51M
        pthread_cond_signal(&ttd->cond);
691
1.51M
        pthread_mutex_unlock(&ttd->lock);
692
1.80M
    found_unlocked:;
693
1.80M
        const int flush = atomic_load(c->flush);
694
1.80M
        int error = atomic_fetch_or(&f->task_thread.error, flush) | flush;
695
696
        // run it
697
1.80M
        tc->f = f;
698
1.80M
        int sby = t->sby;
699
1.80M
        switch (t->type) {
700
64.9k
        case DAV1D_TASK_TYPE_INIT: {
701
64.9k
            assert(c->n_fc > 1);
702
64.9k
            int res = dav1d_decode_frame_init(f);
703
64.9k
            int p1 = f->in_cdf.progress ? atomic_load(f->in_cdf.progress) : 1;
704
64.9k
            if (res || p1 == TILE_ERROR) {
705
101
                pthread_mutex_lock(&ttd->lock);
706
101
                abort_frame(f, res ? res : DAV1D_ERR(EINVAL));
707
101
                reset_task_cur(c, ttd, t->frame_idx);
708
64.8k
            } else {
709
64.8k
                t->type = DAV1D_TASK_TYPE_INIT_CDF;
710
64.8k
                if (p1) goto found_unlocked;
711
12.4k
                add_pending(f, t);
712
12.4k
                pthread_mutex_lock(&ttd->lock);
713
12.4k
            }
714
12.5k
            continue;
715
64.9k
        }
716
64.3k
        case DAV1D_TASK_TYPE_INIT_CDF: {
717
64.3k
            assert(c->n_fc > 1);
718
64.3k
            int res = DAV1D_ERR(EINVAL);
719
64.3k
            if (!atomic_load(&f->task_thread.error))
720
63.5k
                res = dav1d_decode_frame_init_cdf(f);
721
64.3k
            if (f->frame_hdr->refresh_context && !f->task_thread.update_set)
722
64.3k
                atomic_store(f->out_cdf.progress, res < 0 ? TILE_ERROR : 1);
723
190k
            for (int p = 1; p <= 2 && !res; p++)
724
126k
                res = dav1d_task_create_tile_sbrow(f, p, 0);
725
64.3k
            pthread_mutex_lock(&ttd->lock);
726
64.3k
            if (res) {
727
1.04k
                abort_frame(f, DAV1D_ERR(ENOMEM));
728
1.04k
                reset_task_cur(c, ttd, t->frame_idx);
729
1.04k
                atomic_store(&f->task_thread.init_done, 1);
730
1.04k
            }
731
64.3k
            continue;
732
64.3k
        }
733
421k
        case DAV1D_TASK_TYPE_TILE_ENTROPY:
734
841k
        case DAV1D_TASK_TYPE_TILE_RECONSTRUCTION: {
735
841k
            const int p = t->type == DAV1D_TASK_TYPE_TILE_ENTROPY;
736
841k
            const int tile_idx = (int)(t - f->task_thread.tile_tasks[p]);
737
841k
            Dav1dTileState *const ts = &f->ts[tile_idx];
738
739
841k
            tc->ts = ts;
740
841k
            tc->by = sby << f->sb_shift;
741
841k
            const int uses_2pass = c->n_fc > 1;
742
841k
            tc->frame_thread.pass = !uses_2pass ? 0 :
743
841k
                1 + (t->type == DAV1D_TASK_TYPE_TILE_RECONSTRUCTION);
744
841k
            if (!error) error = dav1d_decode_tile_sbrow(tc);
745
841k
            const int progress = error ? TILE_ERROR : 1 + sby;
746
747
            // signal progress
748
841k
            atomic_fetch_or(&f->task_thread.error, error);
749
841k
            if (((sby + 1) << f->sb_shift) < ts->tiling.row_end) {
750
702k
                t->sby++;
751
702k
                t->deps_skip = 0;
752
702k
                if (!check_tile(t, f, uses_2pass)) {
753
248k
                    atomic_store(&ts->progress[p], progress);
754
248k
                    reset_task_cur_async(ttd, t->frame_idx, c->n_fc);
755
248k
                    if (!atomic_fetch_or(&ttd->cond_signaled, 1))
756
100
                        pthread_cond_signal(&ttd->cond);
757
248k
                    goto found_unlocked;
758
248k
                }
759
702k
                atomic_store(&ts->progress[p], progress);
760
454k
                add_pending(f, t);
761
454k
                pthread_mutex_lock(&ttd->lock);
762
454k
            } else {
763
139k
                pthread_mutex_lock(&ttd->lock);
764
139k
                atomic_store(&ts->progress[p], progress);
765
139k
                reset_task_cur(c, ttd, t->frame_idx);
766
139k
                error = atomic_load(&f->task_thread.error);
767
139k
                if (f->frame_hdr->refresh_context &&
768
45.6k
                    tc->frame_thread.pass <= 1 && f->task_thread.update_set &&
769
23.1k
                    f->frame_hdr->tiling.update == tile_idx)
770
22.9k
                {
771
22.9k
                    if (!error)
772
21.9k
                        dav1d_cdf_thread_update(f->frame_hdr, f->out_cdf.data.cdf,
773
21.9k
                                                &f->ts[f->frame_hdr->tiling.update].cdf);
774
22.9k
                    if (c->n_fc > 1)
775
22.9k
                        atomic_store(f->out_cdf.progress, error ? TILE_ERROR : 1);
776
22.9k
                }
777
139k
                if (atomic_fetch_sub(&f->task_thread.task_counter, 1) - 1 == 0 &&
778
139k
                    atomic_load(&f->task_thread.done[0]) &&
779
610
                    (!uses_2pass || atomic_load(&f->task_thread.done[1])))
780
610
                {
781
610
                    error = atomic_load(&f->task_thread.error);
782
610
                    dav1d_decode_frame_exit(f, error == 1 ? DAV1D_ERR(EINVAL) :
783
610
                                            error ? DAV1D_ERR(ENOMEM) : 0);
784
610
                    f->n_tile_data = 0;
785
610
                    pthread_cond_signal(&f->task_thread.cond);
786
610
                }
787
139k
                assert(atomic_load(&f->task_thread.task_counter) >= 0);
788
139k
                if (!atomic_fetch_or(&ttd->cond_signaled, 1))
789
93.8k
                    pthread_cond_signal(&ttd->cond);
790
139k
            }
791
593k
            continue;
792
841k
        }
793
593k
        case DAV1D_TASK_TYPE_DEBLOCK_COLS:
794
255k
            if (!atomic_load(&f->task_thread.error))
795
163k
                f->bd_fn.filter_sbrow_deblock_cols(f, sby);
796
255k
            if (ensure_progress(ttd, f, t, DAV1D_TASK_TYPE_DEBLOCK_ROWS,
797
255k
                                &f->frame_thread.deblock_progress,
798
255k
                                &t->deblock_progress)) continue;
799
            // fall-through
800
342k
        case DAV1D_TASK_TYPE_DEBLOCK_ROWS:
801
342k
            if (!atomic_load(&f->task_thread.error))
802
214k
                f->bd_fn.filter_sbrow_deblock_rows(f, sby);
803
            // signal deblock progress
804
342k
            if (f->frame_hdr->loopfilter.level_y[0] ||
805
113k
                f->frame_hdr->loopfilter.level_y[1])
806
255k
            {
807
255k
                error = atomic_load(&f->task_thread.error);
808
255k
                atomic_store(&f->frame_thread.deblock_progress,
809
255k
                             error ? TILE_ERROR : sby + 1);
810
255k
                reset_task_cur_async(ttd, t->frame_idx, c->n_fc);
811
255k
                if (!atomic_fetch_or(&ttd->cond_signaled, 1))
812
58.0k
                    pthread_cond_signal(&ttd->cond);
813
255k
            } else if (f->seq_hdr->cdef || f->lf.restore_planes) {
814
86.5k
                atomic_fetch_or(&f->frame_thread.copy_lpf_progress[sby >> 5],
815
86.5k
                                1U << (sby & 31));
816
                // CDEF needs the top buffer to be saved by lr_copy_lpf of the
817
                // previous sbrow
818
86.5k
                if (sby) {
819
77.5k
                    int prog = atomic_load(&f->frame_thread.copy_lpf_progress[(sby - 1) >> 5]);
820
77.5k
                    if (~prog & (1U << ((sby - 1) & 31))) {
821
5.68k
                        t->type = DAV1D_TASK_TYPE_CDEF;
822
5.68k
                        t->recon_progress = t->deblock_progress = 0;
823
5.68k
                        add_pending(f, t);
824
5.68k
                        pthread_mutex_lock(&ttd->lock);
825
5.68k
                        continue;
826
5.68k
                    }
827
77.5k
                }
828
86.5k
            }
829
            // fall-through
830
342k
        case DAV1D_TASK_TYPE_CDEF:
831
342k
            if (f->seq_hdr->cdef) {
832
199k
                if (!atomic_load(&f->task_thread.error))
833
130k
                    f->bd_fn.filter_sbrow_cdef(tc, sby);
834
199k
                reset_task_cur_async(ttd, t->frame_idx, c->n_fc);
835
199k
                if (!atomic_fetch_or(&ttd->cond_signaled, 1))
836
38.4k
                    pthread_cond_signal(&ttd->cond);
837
199k
            }
838
            // fall-through
839
344k
        case DAV1D_TASK_TYPE_SUPER_RESOLUTION:
840
344k
            if (f->frame_hdr->width[0] != f->frame_hdr->width[1])
841
25.4k
                if (!atomic_load(&f->task_thread.error))
842
18.9k
                    f->bd_fn.filter_sbrow_resize(f, sby);
843
            // fall-through
844
344k
        case DAV1D_TASK_TYPE_LOOP_RESTORATION:
845
344k
            if (!atomic_load(&f->task_thread.error) && f->lf.restore_planes)
846
97.0k
                f->bd_fn.filter_sbrow_lr(f, sby);
847
            // fall-through
848
412k
        case DAV1D_TASK_TYPE_RECONSTRUCTION_PROGRESS:
849
            // dummy to cover for no post-filters
850
825k
        case DAV1D_TASK_TYPE_ENTROPY_PROGRESS:
851
            // dummy to convert tile progress to frame
852
825k
            break;
853
0
        default: abort();
854
1.80M
        }
855
        // if task completed [typically LR], signal picture progress as per below
856
822k
        const int uses_2pass = c->n_fc > 1;
857
822k
        const int sbh = f->sbh;
858
822k
        const int sbsz = f->sb_step * 4;
859
822k
        if (t->type == DAV1D_TASK_TYPE_ENTROPY_PROGRESS) {
860
412k
            error = atomic_load(&f->task_thread.error);
861
412k
            const unsigned y = sby + 1 == sbh ? UINT_MAX : (unsigned)(sby + 1) * sbsz;
862
412k
            assert(c->n_fc > 1);
863
412k
            if (f->sr_cur.p.data[0] /* upon flush, this can be free'ed already */)
864
412k
                atomic_store(&f->sr_cur.progress[0], error ? FRAME_ERROR : y);
865
412k
            atomic_store(&f->frame_thread.entropy_progress,
866
412k
                         error ? TILE_ERROR : sby + 1);
867
412k
            if (sby + 1 == sbh)
868
412k
                atomic_store(&f->task_thread.done[1], 1);
869
412k
            pthread_mutex_lock(&ttd->lock);
870
412k
            const int num_tasks = atomic_fetch_sub(&f->task_thread.task_counter, 1) - 1;
871
412k
            if (sby + 1 < sbh && num_tasks) {
872
350k
                reset_task_cur(c, ttd, t->frame_idx);
873
350k
                continue;
874
350k
            }
875
62.2k
            if (!num_tasks && atomic_load(&f->task_thread.done[0]) &&
876
62.2k
                atomic_load(&f->task_thread.done[1]))
877
6.87k
            {
878
6.87k
                error = atomic_load(&f->task_thread.error);
879
6.87k
                dav1d_decode_frame_exit(f, error == 1 ? DAV1D_ERR(EINVAL) :
880
6.87k
                                        error ? DAV1D_ERR(ENOMEM) : 0);
881
6.87k
                f->n_tile_data = 0;
882
6.87k
                pthread_cond_signal(&f->task_thread.cond);
883
6.87k
            }
884
62.2k
            reset_task_cur(c, ttd, t->frame_idx);
885
62.2k
            continue;
886
412k
        }
887
    // t->type != DAV1D_TASK_TYPE_ENTROPY_PROGRESS
888
822k
        atomic_fetch_or(&f->frame_thread.frame_progress[sby >> 5],
889
409k
                        1U << (sby & 31));
890
409k
        pthread_mutex_lock(&f->task_thread.lock);
891
409k
        sby = get_frame_progress(c, f);
892
409k
        error = atomic_load(&f->task_thread.error);
893
409k
        const unsigned y = sby + 1 == sbh ? UINT_MAX : (unsigned)(sby + 1) * sbsz;
894
412k
        if (c->n_fc > 1 && f->sr_cur.p.data[0] /* upon flush, this can be free'ed already */)
895
412k
            atomic_store(&f->sr_cur.progress[1], error ? FRAME_ERROR : y);
896
409k
        pthread_mutex_unlock(&f->task_thread.lock);
897
409k
        if (sby + 1 == sbh)
898
409k
            atomic_store(&f->task_thread.done[0], 1);
899
409k
        pthread_mutex_lock(&ttd->lock);
900
409k
        const int num_tasks = atomic_fetch_sub(&f->task_thread.task_counter, 1) - 1;
901
409k
        if (sby + 1 < sbh && num_tasks) {
902
205k
            reset_task_cur(c, ttd, t->frame_idx);
903
205k
            continue;
904
205k
        }
905
204k
        if (!num_tasks && atomic_load(&f->task_thread.done[0]) &&
906
53.5k
            (!uses_2pass || atomic_load(&f->task_thread.done[1])))
907
53.5k
        {
908
53.5k
            error = atomic_load(&f->task_thread.error);
909
53.5k
            dav1d_decode_frame_exit(f, error == 1 ? DAV1D_ERR(EINVAL) :
910
53.5k
                                    error ? DAV1D_ERR(ENOMEM) : 0);
911
53.5k
            f->n_tile_data = 0;
912
53.5k
            pthread_cond_signal(&f->task_thread.cond);
913
53.5k
        }
914
204k
        reset_task_cur(c, ttd, t->frame_idx);
915
204k
    }
916
1.18M
    pthread_mutex_unlock(&ttd->lock);
917
918
    return NULL;
919
1.18M
}