/work/dav1d/src/thread_task.c
Line | Count | Source |
1 | | /* |
2 | | * Copyright © 2018, VideoLAN and dav1d authors |
3 | | * Copyright © 2018, Two Orioles, LLC |
4 | | * All rights reserved. |
5 | | * |
6 | | * Redistribution and use in source and binary forms, with or without |
7 | | * modification, are permitted provided that the following conditions are met: |
8 | | * |
9 | | * 1. Redistributions of source code must retain the above copyright notice, this |
10 | | * list of conditions and the following disclaimer. |
11 | | * |
12 | | * 2. Redistributions in binary form must reproduce the above copyright notice, |
13 | | * this list of conditions and the following disclaimer in the documentation |
14 | | * and/or other materials provided with the distribution. |
15 | | * |
16 | | * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND |
17 | | * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED |
18 | | * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE |
19 | | * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR |
20 | | * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES |
21 | | * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; |
22 | | * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND |
23 | | * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT |
24 | | * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS |
25 | | * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. |
26 | | */ |
27 | | |
28 | | #include "config.h" |
29 | | |
30 | | #include "common/frame.h" |
31 | | |
32 | | #include "src/thread_task.h" |
33 | | #include "src/fg_apply.h" |
34 | | |
35 | | // This function resets the cur pointer to the first frame theoretically |
36 | | // executable after a task completed (ie. each time we update some progress or |
37 | | // insert some tasks in the queue). |
38 | | // When frame_idx is set, it can be either from a completed task, or from tasks |
39 | | // inserted in the queue, in which case we have to make sure the cur pointer |
40 | | // isn't past this insert. |
41 | | // The special case where frame_idx is UINT_MAX is to handle the reset after |
42 | | // completing a task and locklessly signaling progress. In this case we don't |
43 | | // enter a critical section, which is needed for this function, so we set an |
44 | | // atomic for a delayed handling, happening here. Meaning we can call this |
45 | | // function without any actual update other than what's in the atomic, hence |
46 | | // this special case. |
47 | | static inline int reset_task_cur(const Dav1dContext *const c, |
48 | | struct TaskThreadData *const ttd, |
49 | | unsigned frame_idx) |
50 | 16.2M | { |
51 | 16.2M | const unsigned first = atomic_load(&ttd->first); |
52 | 16.2M | unsigned reset_frame_idx = atomic_exchange(&ttd->reset_task_cur, UINT_MAX); |
53 | 16.2M | if (reset_frame_idx < first) { |
54 | 0 | if (frame_idx == UINT_MAX) return 0; |
55 | 0 | reset_frame_idx = UINT_MAX; |
56 | 0 | } |
57 | 16.2M | if (!ttd->cur && c->fc[first].task_thread.task_cur_prev == NULL) |
58 | 2.76M | return 0; |
59 | 13.5M | if (reset_frame_idx != UINT_MAX) { |
60 | 478k | if (frame_idx == UINT_MAX) { |
61 | 255k | if (reset_frame_idx > first + ttd->cur) |
62 | 2.31k | return 0; |
63 | 252k | ttd->cur = reset_frame_idx - first; |
64 | 252k | goto cur_found; |
65 | 255k | } |
66 | 13.0M | } else if (frame_idx == UINT_MAX) |
67 | 10.4M | return 0; |
68 | 2.86M | if (frame_idx < first) frame_idx += c->n_fc; |
69 | 2.86M | const unsigned min_frame_idx = umin(reset_frame_idx, frame_idx); |
70 | 2.86M | const unsigned cur_frame_idx = first + ttd->cur; |
71 | 2.86M | if (ttd->cur < c->n_fc && cur_frame_idx < min_frame_idx) |
72 | 388k | return 0; |
73 | 4.01M | for (ttd->cur = min_frame_idx - first; ttd->cur < c->n_fc; ttd->cur++) |
74 | 3.79M | if (c->fc[(first + ttd->cur) % c->n_fc].task_thread.task_head) |
75 | 2.24M | break; |
76 | 2.72M | cur_found: |
77 | 13.2M | for (unsigned i = ttd->cur; i < c->n_fc; i++) |
78 | 10.4M | c->fc[(first + i) % c->n_fc].task_thread.task_cur_prev = NULL; |
79 | 2.72M | return 1; |
80 | 2.47M | } |
81 | | |
82 | | static inline void reset_task_cur_async(struct TaskThreadData *const ttd, |
83 | | unsigned frame_idx, unsigned n_frames) |
84 | 1.30M | { |
85 | 1.30M | const unsigned first = atomic_load(&ttd->first); |
86 | 1.30M | if (frame_idx < first) frame_idx += n_frames; |
87 | 1.30M | unsigned last_idx = frame_idx; |
88 | 1.30M | do { |
89 | 1.30M | frame_idx = last_idx; |
90 | 1.30M | last_idx = atomic_exchange(&ttd->reset_task_cur, frame_idx); |
91 | 1.30M | } while (last_idx < frame_idx); |
92 | 1.30M | if (frame_idx == first && atomic_load(&ttd->first) != first) { |
93 | 0 | unsigned expected = frame_idx; |
94 | 0 | atomic_compare_exchange_strong(&ttd->reset_task_cur, &expected, UINT_MAX); |
95 | 0 | } |
96 | 1.30M | } |
97 | | |
98 | | static void insert_tasks_between(Dav1dFrameContext *const f, |
99 | | Dav1dTask *const first, Dav1dTask *const last, |
100 | | Dav1dTask *const a, Dav1dTask *const b, |
101 | | const int cond_signal) |
102 | 2.88M | { |
103 | 2.88M | struct TaskThreadData *const ttd = f->task_thread.ttd; |
104 | 2.88M | if (atomic_load(f->c->flush)) return; |
105 | 2.88M | assert(!a || a->next == b); |
106 | 2.88M | if (!a) f->task_thread.task_head = first; |
107 | 2.02M | else a->next = first; |
108 | 2.88M | if (!b) f->task_thread.task_tail = last; |
109 | 2.88M | last->next = b; |
110 | 2.88M | reset_task_cur(f->c, ttd, first->frame_idx); |
111 | 2.88M | if (cond_signal && !atomic_fetch_or(&ttd->cond_signaled, 1)) |
112 | 156k | pthread_cond_signal(&ttd->cond); |
113 | 2.88M | } |
114 | | |
115 | | static void insert_tasks(Dav1dFrameContext *const f, |
116 | | Dav1dTask *const first, Dav1dTask *const last, |
117 | | const int cond_signal) |
118 | 2.88M | { |
119 | | // insert task back into task queue |
120 | 2.88M | Dav1dTask *t_ptr, *prev_t = NULL; |
121 | 2.88M | for (t_ptr = f->task_thread.task_head; |
122 | 8.58M | t_ptr; prev_t = t_ptr, t_ptr = t_ptr->next) |
123 | 6.26M | { |
124 | | // entropy coding precedes other steps |
125 | 6.26M | if (t_ptr->type == DAV1D_TASK_TYPE_TILE_ENTROPY) { |
126 | 1.21M | if (first->type > DAV1D_TASK_TYPE_TILE_ENTROPY) continue; |
127 | | // both are entropy |
128 | 126k | if (first->sby > t_ptr->sby) continue; |
129 | 38.9k | if (first->sby < t_ptr->sby) { |
130 | 1.11k | insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal); |
131 | 1.11k | return; |
132 | 1.11k | } |
133 | | // same sby |
134 | 5.04M | } else { |
135 | 5.04M | if (first->type == DAV1D_TASK_TYPE_TILE_ENTROPY) { |
136 | 288k | insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal); |
137 | 288k | return; |
138 | 288k | } |
139 | 4.75M | if (first->sby > t_ptr->sby) continue; |
140 | 1.56M | if (first->sby < t_ptr->sby) { |
141 | 260k | insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal); |
142 | 260k | return; |
143 | 260k | } |
144 | | // same sby |
145 | 1.30M | if (first->type > t_ptr->type) continue; |
146 | 45.2k | if (first->type < t_ptr->type) { |
147 | 7.43k | insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal); |
148 | 7.43k | return; |
149 | 7.43k | } |
150 | | // same task type |
151 | 45.2k | } |
152 | | |
153 | | // sort by tile-id |
154 | 75.6k | assert(first->type == DAV1D_TASK_TYPE_TILE_RECONSTRUCTION || |
155 | 75.6k | first->type == DAV1D_TASK_TYPE_TILE_ENTROPY); |
156 | 75.6k | assert(first->type == t_ptr->type); |
157 | 75.6k | assert(t_ptr->sby == first->sby); |
158 | 75.6k | const int p = first->type == DAV1D_TASK_TYPE_TILE_ENTROPY; |
159 | 75.6k | const int t_tile_idx = (int) (first - f->task_thread.tile_tasks[p]); |
160 | 75.6k | const int p_tile_idx = (int) (t_ptr - f->task_thread.tile_tasks[p]); |
161 | 75.6k | assert(t_tile_idx != p_tile_idx); |
162 | 75.6k | if (t_tile_idx > p_tile_idx) continue; |
163 | 68 | insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal); |
164 | 68 | return; |
165 | 75.6k | } |
166 | | // append at the end |
167 | 2.32M | insert_tasks_between(f, first, last, prev_t, NULL, cond_signal); |
168 | 2.32M | } |
169 | | |
170 | | static inline void insert_task(Dav1dFrameContext *const f, |
171 | | Dav1dTask *const t, const int cond_signal) |
172 | 2.88M | { |
173 | 2.88M | insert_tasks(f, t, t, cond_signal); |
174 | 2.88M | } |
175 | | |
176 | 601k | static inline void add_pending(Dav1dFrameContext *const f, Dav1dTask *const t) { |
177 | 601k | pthread_mutex_lock(&f->task_thread.pending_tasks.lock); |
178 | 601k | t->next = NULL; |
179 | 601k | if (!f->task_thread.pending_tasks.head) |
180 | 586k | f->task_thread.pending_tasks.head = t; |
181 | 15.2k | else |
182 | 15.2k | f->task_thread.pending_tasks.tail->next = t; |
183 | 601k | f->task_thread.pending_tasks.tail = t; |
184 | 601k | atomic_store(&f->task_thread.pending_tasks.merge, 1); |
185 | 601k | pthread_mutex_unlock(&f->task_thread.pending_tasks.lock); |
186 | 601k | } |
187 | | |
188 | 93.9M | static inline int merge_pending_frame(Dav1dFrameContext *const f) { |
189 | 93.9M | int const merge = atomic_load(&f->task_thread.pending_tasks.merge); |
190 | 93.9M | if (merge) { |
191 | 842k | pthread_mutex_lock(&f->task_thread.pending_tasks.lock); |
192 | 842k | Dav1dTask *t = f->task_thread.pending_tasks.head; |
193 | 842k | f->task_thread.pending_tasks.head = NULL; |
194 | 842k | f->task_thread.pending_tasks.tail = NULL; |
195 | 842k | atomic_store(&f->task_thread.pending_tasks.merge, 0); |
196 | 842k | pthread_mutex_unlock(&f->task_thread.pending_tasks.lock); |
197 | 2.50M | while (t) { |
198 | 1.65M | Dav1dTask *const tmp = t->next; |
199 | 1.65M | insert_task(f, t, 0); |
200 | 1.65M | t = tmp; |
201 | 1.65M | } |
202 | 842k | } |
203 | 93.9M | return merge; |
204 | 93.9M | } |
205 | | |
206 | 14.2M | static inline int merge_pending(const Dav1dContext *const c) { |
207 | 14.2M | int res = 0; |
208 | 99.5M | for (unsigned i = 0; i < c->n_fc; i++) |
209 | 85.3M | res |= merge_pending_frame(&c->fc[i]); |
210 | 14.2M | return res; |
211 | 14.2M | } |
212 | | |
213 | | static int create_filter_sbrow(Dav1dFrameContext *const f, |
214 | | const int pass, Dav1dTask **res_t) |
215 | 516k | { |
216 | 516k | const int has_deblock = f->frame_hdr->loopfilter.level_y[0] || |
217 | 89.4k | f->frame_hdr->loopfilter.level_y[1]; |
218 | 516k | const int has_cdef = f->seq_hdr->cdef; |
219 | 516k | const int has_resize = f->frame_hdr->width[0] != f->frame_hdr->width[1]; |
220 | 516k | const int has_lr = f->lf.restore_planes; |
221 | | |
222 | 516k | Dav1dTask *tasks = f->task_thread.tasks; |
223 | 516k | const int uses_2pass = f->c->n_fc > 1; |
224 | 516k | int num_tasks = f->sbh * (1 + uses_2pass); |
225 | 516k | if (num_tasks > f->task_thread.num_tasks) { |
226 | 124k | const size_t size = sizeof(Dav1dTask) * num_tasks; |
227 | 124k | tasks = dav1d_realloc(ALLOC_COMMON_CTX, f->task_thread.tasks, size); |
228 | 124k | if (!tasks) return -1; |
229 | 124k | memset(tasks, 0, size); |
230 | 124k | f->task_thread.tasks = tasks; |
231 | 124k | f->task_thread.num_tasks = num_tasks; |
232 | 124k | } |
233 | 516k | tasks += f->sbh * (pass & 1); |
234 | | |
235 | 516k | if (pass & 1) { |
236 | 258k | f->frame_thread.entropy_progress = 0; |
237 | 258k | } else { |
238 | 258k | const int prog_sz = ((f->sbh + 31) & ~31) >> 5; |
239 | 258k | if (prog_sz > f->frame_thread.prog_sz) { |
240 | 123k | atomic_uint *const prog = dav1d_realloc(ALLOC_COMMON_CTX, f->frame_thread.frame_progress, |
241 | 123k | 2 * prog_sz * sizeof(*prog)); |
242 | 123k | if (!prog) return -1; |
243 | 123k | f->frame_thread.frame_progress = prog; |
244 | 123k | f->frame_thread.copy_lpf_progress = prog + prog_sz; |
245 | 123k | } |
246 | 258k | f->frame_thread.prog_sz = prog_sz; |
247 | 258k | memset(f->frame_thread.frame_progress, 0, prog_sz * sizeof(atomic_uint)); |
248 | 258k | memset(f->frame_thread.copy_lpf_progress, 0, prog_sz * sizeof(atomic_uint)); |
249 | 258k | atomic_store(&f->frame_thread.deblock_progress, 0); |
250 | 258k | } |
251 | 516k | f->frame_thread.next_tile_row[pass & 1] = 0; |
252 | | |
253 | 516k | Dav1dTask *t = &tasks[0]; |
254 | 516k | t->sby = 0; |
255 | 516k | t->recon_progress = 1; |
256 | 516k | t->deblock_progress = 0; |
257 | 516k | t->type = pass == 1 ? DAV1D_TASK_TYPE_ENTROPY_PROGRESS : |
258 | 516k | has_deblock ? DAV1D_TASK_TYPE_DEBLOCK_COLS : |
259 | 258k | has_cdef || has_lr /* i.e. LR backup */ ? DAV1D_TASK_TYPE_DEBLOCK_ROWS : |
260 | 35.1k | has_resize ? DAV1D_TASK_TYPE_SUPER_RESOLUTION : |
261 | 15.3k | DAV1D_TASK_TYPE_RECONSTRUCTION_PROGRESS; |
262 | 516k | t->frame_idx = (int)(f - f->c->fc); |
263 | | |
264 | 516k | *res_t = t; |
265 | 516k | return 0; |
266 | 516k | } |
267 | | |
268 | | int dav1d_task_create_tile_sbrow(Dav1dFrameContext *const f, const int pass, |
269 | | const int cond_signal) |
270 | 516k | { |
271 | 516k | Dav1dTask *tasks = f->task_thread.tile_tasks[0]; |
272 | 516k | const int uses_2pass = f->c->n_fc > 1; |
273 | 516k | const int n_tasks_per_pass = f->frame_hdr->tiling.cols * f->frame_hdr->tiling.rows; |
274 | 516k | const int n_tasks = n_tasks_per_pass * (1 + uses_2pass); |
275 | 516k | if (pass < 2) { |
276 | 258k | if (n_tasks > f->task_thread.num_tile_tasks) { |
277 | 123k | const size_t size = sizeof(Dav1dTask) * n_tasks; |
278 | 123k | tasks = dav1d_realloc(ALLOC_COMMON_CTX, f->task_thread.tile_tasks[0], size); |
279 | 123k | if (!tasks) return -1; |
280 | 123k | memset(tasks, 0, size); |
281 | 123k | f->task_thread.tile_tasks[0] = tasks; |
282 | 123k | f->task_thread.num_tile_tasks = n_tasks; |
283 | 123k | } |
284 | 258k | f->task_thread.tile_tasks[1] = tasks + n_tasks_per_pass; |
285 | 258k | } |
286 | 516k | assert(n_tasks <= f->task_thread.num_tile_tasks); |
287 | | |
288 | 516k | Dav1dTask *pf_t; |
289 | 516k | if (create_filter_sbrow(f, pass, &pf_t)) |
290 | 0 | return -1; |
291 | | |
292 | 516k | Dav1dTask *const p1_tasks = f->task_thread.tile_tasks[1]; |
293 | 516k | Dav1dTask *prev_t = NULL; |
294 | 516k | if (pass == 2) { |
295 | 258k | prev_t = &p1_tasks[n_tasks_per_pass - 1]; |
296 | | // PF task is scheduled after the last sby=0 TILE task |
297 | 258k | if (f->frame_hdr->tiling.rows == 1) |
298 | 255k | prev_t = prev_t->next; |
299 | 258k | } |
300 | 516k | tasks += (pass & 1) * n_tasks_per_pass; |
301 | 1.05M | for (int tile_idx = 0; tile_idx < n_tasks_per_pass; tile_idx++) { |
302 | 543k | Dav1dTileState *const ts = &f->ts[tile_idx]; |
303 | 543k | Dav1dTask *t = &tasks[tile_idx]; |
304 | 543k | t->sby = ts->tiling.row_start >> f->sb_shift; |
305 | 543k | if (pf_t && t->sby) { |
306 | 5.51k | prev_t->next = pf_t; |
307 | 5.51k | prev_t = pf_t; |
308 | 5.51k | pf_t = NULL; |
309 | 5.51k | } |
310 | 543k | t->recon_progress = 0; |
311 | 543k | t->deblock_progress = 0; |
312 | 543k | t->deps_skip = 0; |
313 | 543k | t->type = pass != 1 ? DAV1D_TASK_TYPE_TILE_RECONSTRUCTION : |
314 | 543k | DAV1D_TASK_TYPE_TILE_ENTROPY; |
315 | 543k | t->frame_idx = (int)(f - f->c->fc); |
316 | 543k | if (prev_t) prev_t->next = t; |
317 | 543k | prev_t = t; |
318 | 543k | } |
319 | 516k | if (pf_t) { |
320 | 510k | prev_t->next = pf_t; |
321 | 510k | prev_t = pf_t; |
322 | 510k | } |
323 | 516k | prev_t->next = NULL; |
324 | | |
325 | 516k | atomic_store(&f->task_thread.done[pass & 1], 0); |
326 | | |
327 | | // XXX in theory this could be done locklessly, at this point they are no |
328 | | // tasks in the frameQ, so no other runner should be using this lock, but |
329 | | // we must add both passes at once |
330 | 516k | if (!(pass & 1)) { |
331 | 258k | pthread_mutex_lock(&f->task_thread.pending_tasks.lock); |
332 | 258k | assert(f->task_thread.pending_tasks.head == NULL); |
333 | 258k | f->task_thread.pending_tasks.head = f->task_thread.tile_tasks[pass == 2]; |
334 | 258k | f->task_thread.pending_tasks.tail = prev_t; |
335 | 258k | atomic_store(&f->task_thread.pending_tasks.merge, 1); |
336 | 258k | atomic_store(&f->task_thread.init_done, 1); |
337 | 258k | pthread_mutex_unlock(&f->task_thread.pending_tasks.lock); |
338 | 258k | } |
339 | 516k | return 0; |
340 | 516k | } |
341 | | |
342 | 266k | void dav1d_task_frame_init(Dav1dFrameContext *const f) { |
343 | 266k | const Dav1dContext *const c = f->c; |
344 | | |
345 | 266k | atomic_store(&f->task_thread.init_done, 0); |
346 | | // schedule init task, which will schedule the remaining tasks |
347 | 266k | Dav1dTask *const t = &f->task_thread.init_task; |
348 | 266k | t->type = DAV1D_TASK_TYPE_INIT; |
349 | 266k | t->frame_idx = (int)(f - c->fc); |
350 | 266k | t->sby = 0; |
351 | 266k | t->recon_progress = t->deblock_progress = 0; |
352 | 266k | insert_task(f, t, 1); |
353 | 266k | } |
354 | | |
355 | | void dav1d_task_delayed_fg(Dav1dContext *const c, Dav1dPicture *const out, |
356 | | const Dav1dPicture *const in) |
357 | 4.90k | { |
358 | 4.90k | struct TaskThreadData *const ttd = &c->task_thread; |
359 | 4.90k | ttd->delayed_fg.in = in; |
360 | 4.90k | ttd->delayed_fg.out = out; |
361 | 4.90k | ttd->delayed_fg.type = DAV1D_TASK_TYPE_FG_PREP; |
362 | 4.90k | atomic_init(&ttd->delayed_fg.progress[0], 0); |
363 | 4.90k | atomic_init(&ttd->delayed_fg.progress[1], 0); |
364 | 4.90k | pthread_mutex_lock(&ttd->lock); |
365 | 4.90k | ttd->delayed_fg.exec = 1; |
366 | 4.90k | ttd->delayed_fg.finished = 0; |
367 | 4.90k | pthread_cond_signal(&ttd->cond); |
368 | 4.90k | do { |
369 | 4.90k | pthread_cond_wait(&ttd->delayed_fg.cond, &ttd->lock); |
370 | 4.90k | } while (!ttd->delayed_fg.finished); |
371 | 4.90k | pthread_mutex_unlock(&ttd->lock); |
372 | 4.90k | } |
373 | | |
374 | | static inline int ensure_progress(struct TaskThreadData *const ttd, |
375 | | Dav1dFrameContext *const f, |
376 | | Dav1dTask *const t, const enum TaskType type, |
377 | | atomic_int *const state, int *const target) |
378 | 482k | { |
379 | | // deblock_rows (non-LR portion) depends on deblock of previous sbrow, |
380 | | // so ensure that completed. if not, re-add to task-queue; else, fall-through |
381 | 482k | int p1 = atomic_load(state); |
382 | 482k | if (p1 < t->sby) { |
383 | 13.8k | t->type = type; |
384 | 13.8k | t->recon_progress = t->deblock_progress = 0; |
385 | 13.8k | *target = t->sby; |
386 | 13.8k | add_pending(f, t); |
387 | 13.8k | pthread_mutex_lock(&ttd->lock); |
388 | 13.8k | return 1; |
389 | 13.8k | } |
390 | 468k | return 0; |
391 | 482k | } |
392 | | |
393 | | static inline int check_tile(Dav1dTask *const t, Dav1dFrameContext *const f, |
394 | | const int frame_mt) |
395 | 3.75M | { |
396 | 3.75M | const int tp = t->type == DAV1D_TASK_TYPE_TILE_ENTROPY; |
397 | 3.75M | const int tile_idx = (int)(t - f->task_thread.tile_tasks[tp]); |
398 | 3.75M | Dav1dTileState *const ts = &f->ts[tile_idx]; |
399 | 3.75M | const int p1 = atomic_load(&ts->progress[tp]); |
400 | 3.75M | if (p1 < t->sby) return 1; |
401 | 3.21M | int error = p1 == TILE_ERROR; |
402 | 3.21M | error |= atomic_fetch_or(&f->task_thread.error, error); |
403 | 3.21M | if (!error && frame_mt && !tp) { |
404 | 1.82M | const int p2 = atomic_load(&ts->progress[1]); |
405 | 1.82M | if (p2 <= t->sby) return 1; |
406 | 1.11M | error = p2 == TILE_ERROR; |
407 | 1.11M | error |= atomic_fetch_or(&f->task_thread.error, error); |
408 | 1.11M | } |
409 | 2.50M | if (!error && frame_mt && !IS_KEY_OR_INTRA(f->frame_hdr)) { |
410 | | // check reference state |
411 | 1.47M | const Dav1dThreadPicture *p = &f->sr_cur; |
412 | 1.47M | const int ss_ver = p->p.p.layout == DAV1D_PIXEL_LAYOUT_I420; |
413 | 1.47M | const unsigned p_b = (t->sby + 1) << (f->sb_shift + 2); |
414 | 1.47M | const int tile_sby = t->sby - (ts->tiling.row_start >> f->sb_shift); |
415 | 1.47M | const int (*const lowest_px)[2] = ts->lowest_pixel[tile_sby]; |
416 | 4.71M | for (int n = t->deps_skip; n < 7; n++, t->deps_skip++) { |
417 | 4.25M | unsigned lowest; |
418 | 4.25M | if (tp) { |
419 | | // if temporal mv refs are disabled, we only need this |
420 | | // for the primary ref; if segmentation is disabled, we |
421 | | // don't even need that |
422 | 2.03M | lowest = p_b; |
423 | 2.21M | } else { |
424 | | // +8 is postfilter-induced delay |
425 | 2.21M | const int y = lowest_px[n][0] == INT_MIN ? INT_MIN : |
426 | 2.21M | lowest_px[n][0] + 8; |
427 | 2.21M | const int uv = lowest_px[n][1] == INT_MIN ? INT_MIN : |
428 | 2.21M | lowest_px[n][1] * (1 << ss_ver) + 8; |
429 | 2.21M | const int max = imax(y, uv); |
430 | 2.21M | if (max == INT_MIN) continue; |
431 | 963k | lowest = iclip(max, 1, f->refp[n].p.p.h); |
432 | 963k | } |
433 | 3.00M | const unsigned p3 = atomic_load(&f->refp[n].progress[!tp]); |
434 | 3.00M | if (p3 < lowest) return 1; |
435 | 3.00M | atomic_fetch_or(&f->task_thread.error, p3 == FRAME_ERROR); |
436 | 1.98M | } |
437 | 1.47M | } |
438 | 1.49M | return 0; |
439 | 2.50M | } |
440 | | |
441 | | static inline int get_frame_progress(const Dav1dContext *const c, |
442 | | const Dav1dFrameContext *const f) |
443 | 726k | { |
444 | 726k | unsigned frame_prog = c->n_fc > 1 ? atomic_load(&f->sr_cur.progress[1]) : 0; |
445 | 726k | if (frame_prog >= FRAME_ERROR) |
446 | 248k | return f->sbh - 1; |
447 | 478k | int idx = frame_prog >> (f->sb_shift + 7); |
448 | 478k | int prog; |
449 | 481k | do { |
450 | 481k | atomic_uint *state = &f->frame_thread.frame_progress[idx]; |
451 | 481k | const unsigned val = ~atomic_load(state); |
452 | 481k | prog = val ? ctz(val) : 32; |
453 | 481k | if (prog != 32) break; |
454 | 2.82k | prog = 0; |
455 | 2.82k | } while (++idx < f->frame_thread.prog_sz); |
456 | 478k | return ((idx << 5) | prog) - 1; |
457 | 726k | } |
458 | | |
459 | 5.62k | static inline void abort_frame(Dav1dFrameContext *const f, const int error) { |
460 | 5.62k | atomic_store(&f->task_thread.error, error == DAV1D_ERR(EINVAL) ? 1 : -1); |
461 | 5.62k | atomic_store(&f->task_thread.task_counter, 0); |
462 | 5.62k | atomic_store(&f->task_thread.done[0], 1); |
463 | 5.62k | atomic_store(&f->task_thread.done[1], 1); |
464 | 5.62k | atomic_store(&f->sr_cur.progress[0], FRAME_ERROR); |
465 | 5.62k | atomic_store(&f->sr_cur.progress[1], FRAME_ERROR); |
466 | 5.62k | dav1d_decode_frame_exit(f, error); |
467 | 5.62k | f->n_tile_data = 0; |
468 | 5.62k | pthread_cond_signal(&f->task_thread.cond); |
469 | 5.62k | } |
470 | | |
471 | | static inline void delayed_fg_task(const Dav1dContext *const c, |
472 | | struct TaskThreadData *const ttd) |
473 | 27.8k | { |
474 | 27.8k | const Dav1dPicture *const in = ttd->delayed_fg.in; |
475 | 27.8k | Dav1dPicture *const out = ttd->delayed_fg.out; |
476 | 27.8k | #if CONFIG_16BPC |
477 | 27.8k | int off; |
478 | 27.8k | if (out->p.bpc != 8) |
479 | 14.0k | off = (out->p.bpc >> 1) - 4; |
480 | 27.8k | #endif |
481 | 27.8k | switch (ttd->delayed_fg.type) { |
482 | 4.90k | case DAV1D_TASK_TYPE_FG_PREP: |
483 | 4.90k | ttd->delayed_fg.exec = 0; |
484 | 4.90k | if (atomic_load(&ttd->cond_signaled)) |
485 | 8 | pthread_cond_signal(&ttd->cond); |
486 | 4.90k | pthread_mutex_unlock(&ttd->lock); |
487 | 4.90k | switch (out->p.bpc) { |
488 | 0 | #if CONFIG_8BPC |
489 | 2.25k | case 8: |
490 | 2.25k | dav1d_prep_grain_8bpc(&c->dsp[0].fg, out, in, |
491 | 2.25k | ttd->delayed_fg.scaling_8bpc, |
492 | 2.25k | ttd->delayed_fg.grain_lut_8bpc); |
493 | 2.25k | break; |
494 | 0 | #endif |
495 | 0 | #if CONFIG_16BPC |
496 | 2.39k | case 10: |
497 | 2.65k | case 12: |
498 | 2.65k | dav1d_prep_grain_16bpc(&c->dsp[off].fg, out, in, |
499 | 2.65k | ttd->delayed_fg.scaling_16bpc, |
500 | 2.65k | ttd->delayed_fg.grain_lut_16bpc); |
501 | 2.65k | break; |
502 | 0 | #endif |
503 | 0 | default: abort(); |
504 | 4.90k | } |
505 | 4.90k | ttd->delayed_fg.type = DAV1D_TASK_TYPE_FG_APPLY; |
506 | 4.90k | pthread_mutex_lock(&ttd->lock); |
507 | 4.90k | ttd->delayed_fg.exec = 1; |
508 | | // fall-through |
509 | 27.8k | case DAV1D_TASK_TYPE_FG_APPLY:; |
510 | 27.8k | int row = atomic_fetch_add(&ttd->delayed_fg.progress[0], 1); |
511 | 27.8k | pthread_mutex_unlock(&ttd->lock); |
512 | 27.8k | int progmax = (out->p.h + FG_BLOCK_SIZE - 1) / FG_BLOCK_SIZE; |
513 | 76.9k | while (row < progmax) { |
514 | 52.7k | if (row + 1 < progmax) |
515 | 47.8k | pthread_cond_signal(&ttd->cond); |
516 | 4.89k | else { |
517 | 4.89k | pthread_mutex_lock(&ttd->lock); |
518 | 4.89k | ttd->delayed_fg.exec = 0; |
519 | 4.89k | pthread_mutex_unlock(&ttd->lock); |
520 | 4.89k | } |
521 | 52.7k | switch (out->p.bpc) { |
522 | 0 | #if CONFIG_8BPC |
523 | 25.9k | case 8: |
524 | 25.9k | dav1d_apply_grain_row_8bpc(&c->dsp[0].fg, out, in, |
525 | 25.9k | ttd->delayed_fg.scaling_8bpc, |
526 | 25.9k | ttd->delayed_fg.grain_lut_8bpc, row); |
527 | 25.9k | break; |
528 | 0 | #endif |
529 | 0 | #if CONFIG_16BPC |
530 | 24.7k | case 10: |
531 | 26.7k | case 12: |
532 | 26.7k | dav1d_apply_grain_row_16bpc(&c->dsp[off].fg, out, in, |
533 | 26.7k | ttd->delayed_fg.scaling_16bpc, |
534 | 26.7k | ttd->delayed_fg.grain_lut_16bpc, row); |
535 | 26.7k | break; |
536 | 0 | #endif |
537 | 0 | default: abort(); |
538 | 52.7k | } |
539 | 49.1k | row = atomic_fetch_add(&ttd->delayed_fg.progress[0], 1); |
540 | 49.1k | atomic_fetch_add(&ttd->delayed_fg.progress[1], 1); |
541 | 49.1k | } |
542 | 24.2k | pthread_mutex_lock(&ttd->lock); |
543 | 24.2k | ttd->delayed_fg.exec = 0; |
544 | 24.2k | int done = atomic_fetch_add(&ttd->delayed_fg.progress[1], 1) + 1; |
545 | 24.2k | progmax = atomic_load(&ttd->delayed_fg.progress[0]); |
546 | | // signal for completion only once the last runner reaches this |
547 | 24.2k | if (done >= progmax) { |
548 | 4.90k | ttd->delayed_fg.finished = 1; |
549 | 4.90k | pthread_cond_signal(&ttd->delayed_fg.cond); |
550 | 4.90k | } |
551 | 24.2k | break; |
552 | 0 | default: abort(); |
553 | 27.8k | } |
554 | 27.8k | } |
555 | | |
556 | 2.74M | void *dav1d_worker_task(void *data) { |
557 | 2.74M | Dav1dTaskContext *const tc = data; |
558 | 2.74M | const Dav1dContext *const c = tc->c; |
559 | 2.74M | struct TaskThreadData *const ttd = tc->task_thread.ttd; |
560 | | |
561 | 2.74M | dav1d_set_thread_name("dav1d-worker"); |
562 | | |
563 | 2.74M | pthread_mutex_lock(&ttd->lock); |
564 | 11.3M | for (;;) { |
565 | 11.3M | if (tc->task_thread.die) break; |
566 | 8.64M | if (atomic_load(c->flush)) goto park; |
567 | | |
568 | 8.57M | merge_pending(c); |
569 | 8.57M | if (ttd->delayed_fg.exec) { // run delayed film grain first |
570 | 27.8k | delayed_fg_task(c, ttd); |
571 | 27.8k | continue; |
572 | 27.8k | } |
573 | 8.54M | Dav1dFrameContext *f; |
574 | 8.54M | Dav1dTask *t, *prev_t = NULL; |
575 | 8.54M | if (c->n_fc > 1) { // run init tasks second |
576 | 58.8M | for (unsigned i = 0; i < c->n_fc; i++) { |
577 | 50.6M | const unsigned first = atomic_load(&ttd->first); |
578 | 50.6M | f = &c->fc[(first + i) % c->n_fc]; |
579 | 50.6M | if (atomic_load(&f->task_thread.init_done)) continue; |
580 | 33.2M | t = f->task_thread.task_head; |
581 | 33.2M | if (!t) continue; |
582 | 997k | if (t->type == DAV1D_TASK_TYPE_INIT) goto found; |
583 | 732k | if (t->type == DAV1D_TASK_TYPE_INIT_CDF) { |
584 | | // XXX This can be a simple else, if adding tasks of both |
585 | | // passes at once (in dav1d_task_create_tile_sbrow). |
586 | | // Adding the tasks to the pending Q can result in a |
587 | | // thread merging them before setting init_done. |
588 | | // We will need to set init_done before adding to the |
589 | | // pending Q, so maybe return the tasks, set init_done, |
590 | | // and add to pending Q only then. |
591 | 732k | const int p1 = f->in_cdf.progress ? |
592 | 732k | atomic_load(f->in_cdf.progress) : 1; |
593 | 732k | if (p1) { |
594 | 40.6k | atomic_fetch_or(&f->task_thread.error, p1 == TILE_ERROR); |
595 | 40.6k | goto found; |
596 | 40.6k | } |
597 | 732k | } |
598 | 732k | } |
599 | 8.54M | } |
600 | 14.3M | while (ttd->cur < c->n_fc) { // run decoding tasks last |
601 | 8.61M | const unsigned first = atomic_load(&ttd->first); |
602 | 8.61M | f = &c->fc[(first + ttd->cur) % c->n_fc]; |
603 | 8.61M | merge_pending_frame(f); |
604 | 8.61M | prev_t = f->task_thread.task_cur_prev; |
605 | 8.61M | t = prev_t ? prev_t->next : f->task_thread.task_head; |
606 | 15.3M | while (t) { |
607 | 9.33M | if (t->type == DAV1D_TASK_TYPE_INIT_CDF) goto next; |
608 | 9.16M | else if (t->type == DAV1D_TASK_TYPE_TILE_ENTROPY || |
609 | 8.24M | t->type == DAV1D_TASK_TYPE_TILE_RECONSTRUCTION) |
610 | 2.79M | { |
611 | | // if not bottom sbrow of tile, this task will be re-added |
612 | | // after it's finished |
613 | 2.79M | if (!check_tile(t, f, c->n_fc > 1)) |
614 | 1.06M | goto found; |
615 | 6.37M | } else if (t->recon_progress) { |
616 | 4.82M | const int p = t->type == DAV1D_TASK_TYPE_ENTROPY_PROGRESS; |
617 | 4.82M | int error = atomic_load(&f->task_thread.error); |
618 | 4.82M | assert(!atomic_load(&f->task_thread.done[p]) || error); |
619 | 4.82M | const int tile_row_base = f->frame_hdr->tiling.cols * |
620 | 4.82M | f->frame_thread.next_tile_row[p]; |
621 | 4.82M | if (p) { |
622 | 1.75M | atomic_int *const prog = &f->frame_thread.entropy_progress; |
623 | 1.75M | const int p1 = atomic_load(prog); |
624 | 1.75M | if (p1 < t->sby) goto next; |
625 | 1.75M | atomic_fetch_or(&f->task_thread.error, p1 == TILE_ERROR); |
626 | 1.65M | } |
627 | 6.24M | for (int tc = 0; tc < f->frame_hdr->tiling.cols; tc++) { |
628 | 4.78M | Dav1dTileState *const ts = &f->ts[tile_row_base + tc]; |
629 | 4.78M | const int p2 = atomic_load(&ts->progress[p]); |
630 | 4.78M | if (p2 < t->recon_progress) goto next; |
631 | 4.78M | atomic_fetch_or(&f->task_thread.error, p2 == TILE_ERROR); |
632 | 1.51M | } |
633 | 1.45M | if (t->sby + 1 < f->sbh) { |
634 | | // add sby+1 to list to replace this one |
635 | 959k | Dav1dTask *next_t = &t[1]; |
636 | 959k | *next_t = *t; |
637 | 959k | next_t->sby++; |
638 | 959k | const int ntr = f->frame_thread.next_tile_row[p] + 1; |
639 | 959k | const int start = f->frame_hdr->tiling.row_start_sb[ntr]; |
640 | 959k | if (next_t->sby == start) |
641 | 12.0k | f->frame_thread.next_tile_row[p] = ntr; |
642 | 959k | next_t->recon_progress = next_t->sby + 1; |
643 | 959k | insert_task(f, next_t, 0); |
644 | 959k | } |
645 | 1.45M | goto found; |
646 | 4.72M | } else if (t->type == DAV1D_TASK_TYPE_CDEF) { |
647 | 87.4k | atomic_uint *prog = f->frame_thread.copy_lpf_progress; |
648 | 87.4k | const int p1 = atomic_load(&prog[(t->sby - 1) >> 5]); |
649 | 87.4k | if (p1 & (1U << ((t->sby - 1) & 31))) |
650 | 14.6k | goto found; |
651 | 1.46M | } else { |
652 | 1.46M | assert(t->deblock_progress); |
653 | 1.46M | const int p1 = atomic_load(&f->frame_thread.deblock_progress); |
654 | 1.46M | if (p1 >= t->deblock_progress) { |
655 | 13.8k | atomic_fetch_or(&f->task_thread.error, p1 == TILE_ERROR); |
656 | 13.8k | goto found; |
657 | 13.8k | } |
658 | 1.46M | } |
659 | 6.77M | next: |
660 | 6.77M | prev_t = t; |
661 | 6.77M | t = t->next; |
662 | 6.77M | f->task_thread.task_cur_prev = prev_t; |
663 | 6.77M | } |
664 | 6.06M | ttd->cur++; |
665 | 6.06M | } |
666 | 5.68M | if (reset_task_cur(c, ttd, UINT_MAX)) continue; |
667 | 5.65M | if (merge_pending(c)) continue; |
668 | 5.72M | park: |
669 | 5.72M | tc->task_thread.flushed = 1; |
670 | 5.72M | pthread_cond_signal(&tc->task_thread.td.cond); |
671 | | // we want to be woken up next time progress is signaled |
672 | 5.72M | atomic_store(&ttd->cond_signaled, 0); |
673 | 5.72M | pthread_cond_wait(&ttd->cond, &ttd->lock); |
674 | 5.72M | tc->task_thread.flushed = 0; |
675 | 5.72M | reset_task_cur(c, ttd, UINT_MAX); |
676 | 5.72M | continue; |
677 | | |
678 | 2.85M | found: |
679 | | // remove t from list |
680 | 2.85M | if (prev_t) prev_t->next = t->next; |
681 | 2.47M | else f->task_thread.task_head = t->next; |
682 | 2.85M | if (!t->next) f->task_thread.task_tail = prev_t; |
683 | 2.85M | if (t->type > DAV1D_TASK_TYPE_INIT_CDF && !f->task_thread.task_head) |
684 | 248k | ttd->cur++; |
685 | 2.85M | t->next = NULL; |
686 | | // we don't need to check cond_signaled here, since we found a task |
687 | | // after the last signal so we want to re-signal the next waiting thread |
688 | | // and again won't need to signal after that |
689 | 2.85M | atomic_store(&ttd->cond_signaled, 1); |
690 | 2.85M | pthread_cond_signal(&ttd->cond); |
691 | 2.85M | pthread_mutex_unlock(&ttd->lock); |
692 | 3.50M | found_unlocked:; |
693 | 3.50M | const int flush = atomic_load(c->flush); |
694 | 3.50M | int error = atomic_fetch_or(&f->task_thread.error, flush) | flush; |
695 | | |
696 | | // run it |
697 | 3.50M | tc->f = f; |
698 | 3.50M | int sby = t->sby; |
699 | 3.50M | switch (t->type) { |
700 | 264k | case DAV1D_TASK_TYPE_INIT: { |
701 | 264k | assert(c->n_fc > 1); |
702 | 264k | int res = dav1d_decode_frame_init(f); |
703 | 264k | int p1 = f->in_cdf.progress ? atomic_load(f->in_cdf.progress) : 1; |
704 | 264k | if (res || p1 == TILE_ERROR) { |
705 | 365 | pthread_mutex_lock(&ttd->lock); |
706 | 365 | abort_frame(f, res ? res : DAV1D_ERR(EINVAL)); |
707 | 365 | reset_task_cur(c, ttd, t->frame_idx); |
708 | 264k | } else { |
709 | 264k | t->type = DAV1D_TASK_TYPE_INIT_CDF; |
710 | 264k | if (p1) goto found_unlocked; |
711 | 41.3k | add_pending(f, t); |
712 | 41.3k | pthread_mutex_lock(&ttd->lock); |
713 | 41.3k | } |
714 | 41.7k | continue; |
715 | 264k | } |
716 | 263k | case DAV1D_TASK_TYPE_INIT_CDF: { |
717 | 263k | assert(c->n_fc > 1); |
718 | 263k | int res = DAV1D_ERR(EINVAL); |
719 | 263k | if (!atomic_load(&f->task_thread.error)) |
720 | 259k | res = dav1d_decode_frame_init_cdf(f); |
721 | 263k | if (f->frame_hdr->refresh_context && !f->task_thread.update_set) |
722 | 263k | atomic_store(f->out_cdf.progress, res < 0 ? TILE_ERROR : 1); |
723 | 779k | for (int p = 1; p <= 2 && !res; p++) |
724 | 516k | res = dav1d_task_create_tile_sbrow(f, p, 0); |
725 | 263k | pthread_mutex_lock(&ttd->lock); |
726 | 263k | if (res) { |
727 | 5.25k | abort_frame(f, DAV1D_ERR(ENOMEM)); |
728 | 5.25k | reset_task_cur(c, ttd, t->frame_idx); |
729 | 5.25k | atomic_store(&f->task_thread.init_done, 1); |
730 | 5.25k | } |
731 | 263k | continue; |
732 | 263k | } |
733 | 751k | case DAV1D_TASK_TYPE_TILE_ENTROPY: |
734 | 1.49M | case DAV1D_TASK_TYPE_TILE_RECONSTRUCTION: { |
735 | 1.49M | const int p = t->type == DAV1D_TASK_TYPE_TILE_ENTROPY; |
736 | 1.49M | const int tile_idx = (int)(t - f->task_thread.tile_tasks[p]); |
737 | 1.49M | Dav1dTileState *const ts = &f->ts[tile_idx]; |
738 | | |
739 | 1.49M | tc->ts = ts; |
740 | 1.49M | tc->by = sby << f->sb_shift; |
741 | 1.49M | const int uses_2pass = c->n_fc > 1; |
742 | 1.49M | tc->frame_thread.pass = !uses_2pass ? 0 : |
743 | 1.49M | 1 + (t->type == DAV1D_TASK_TYPE_TILE_RECONSTRUCTION); |
744 | 1.49M | if (!error) error = dav1d_decode_tile_sbrow(tc); |
745 | 1.49M | const int progress = error ? TILE_ERROR : 1 + sby; |
746 | | |
747 | | // signal progress |
748 | 1.49M | atomic_fetch_or(&f->task_thread.error, error); |
749 | 1.49M | if (((sby + 1) << f->sb_shift) < ts->tiling.row_end) { |
750 | 961k | t->sby++; |
751 | 961k | t->deps_skip = 0; |
752 | 961k | if (!check_tile(t, f, uses_2pass)) { |
753 | 429k | atomic_store(&ts->progress[p], progress); |
754 | 429k | reset_task_cur_async(ttd, t->frame_idx, c->n_fc); |
755 | 429k | if (!atomic_fetch_or(&ttd->cond_signaled, 1)) |
756 | 145 | pthread_cond_signal(&ttd->cond); |
757 | 429k | goto found_unlocked; |
758 | 429k | } |
759 | 961k | atomic_store(&ts->progress[p], progress); |
760 | 532k | add_pending(f, t); |
761 | 532k | pthread_mutex_lock(&ttd->lock); |
762 | 534k | } else { |
763 | 534k | pthread_mutex_lock(&ttd->lock); |
764 | 534k | atomic_store(&ts->progress[p], progress); |
765 | 534k | reset_task_cur(c, ttd, t->frame_idx); |
766 | 534k | error = atomic_load(&f->task_thread.error); |
767 | 534k | if (f->frame_hdr->refresh_context && |
768 | 298k | tc->frame_thread.pass <= 1 && f->task_thread.update_set && |
769 | 150k | f->frame_hdr->tiling.update == tile_idx) |
770 | 149k | { |
771 | 149k | if (!error) |
772 | 144k | dav1d_cdf_thread_update(f->frame_hdr, f->out_cdf.data.cdf, |
773 | 144k | &f->ts[f->frame_hdr->tiling.update].cdf); |
774 | 149k | if (c->n_fc > 1) |
775 | 149k | atomic_store(f->out_cdf.progress, error ? TILE_ERROR : 1); |
776 | 149k | } |
777 | 534k | if (atomic_fetch_sub(&f->task_thread.task_counter, 1) - 1 == 0 && |
778 | 534k | atomic_load(&f->task_thread.done[0]) && |
779 | 1.74k | (!uses_2pass || atomic_load(&f->task_thread.done[1]))) |
780 | 1.74k | { |
781 | 1.74k | error = atomic_load(&f->task_thread.error); |
782 | 1.74k | dav1d_decode_frame_exit(f, error == 1 ? DAV1D_ERR(EINVAL) : |
783 | 1.74k | error ? DAV1D_ERR(ENOMEM) : 0); |
784 | 1.74k | f->n_tile_data = 0; |
785 | 1.74k | pthread_cond_signal(&f->task_thread.cond); |
786 | 1.74k | } |
787 | 534k | assert(atomic_load(&f->task_thread.task_counter) >= 0); |
788 | 534k | if (!atomic_fetch_or(&ttd->cond_signaled, 1)) |
789 | 356k | pthread_cond_signal(&ttd->cond); |
790 | 534k | } |
791 | 1.06M | continue; |
792 | 1.49M | } |
793 | 1.06M | case DAV1D_TASK_TYPE_DEBLOCK_COLS: |
794 | 482k | if (!atomic_load(&f->task_thread.error)) |
795 | 360k | f->bd_fn.filter_sbrow_deblock_cols(f, sby); |
796 | 482k | if (ensure_progress(ttd, f, t, DAV1D_TASK_TYPE_DEBLOCK_ROWS, |
797 | 482k | &f->frame_thread.deblock_progress, |
798 | 482k | &t->deblock_progress)) continue; |
799 | | // fall-through |
800 | 633k | case DAV1D_TASK_TYPE_DEBLOCK_ROWS: |
801 | 633k | if (!atomic_load(&f->task_thread.error)) |
802 | 421k | f->bd_fn.filter_sbrow_deblock_rows(f, sby); |
803 | | // signal deblock progress |
804 | 633k | if (f->frame_hdr->loopfilter.level_y[0] || |
805 | 183k | f->frame_hdr->loopfilter.level_y[1]) |
806 | 482k | { |
807 | 482k | error = atomic_load(&f->task_thread.error); |
808 | 482k | atomic_store(&f->frame_thread.deblock_progress, |
809 | 482k | error ? TILE_ERROR : sby + 1); |
810 | 482k | reset_task_cur_async(ttd, t->frame_idx, c->n_fc); |
811 | 482k | if (!atomic_fetch_or(&ttd->cond_signaled, 1)) |
812 | 133k | pthread_cond_signal(&ttd->cond); |
813 | 482k | } else if (f->seq_hdr->cdef || f->lf.restore_planes) { |
814 | 151k | atomic_fetch_or(&f->frame_thread.copy_lpf_progress[sby >> 5], |
815 | 151k | 1U << (sby & 31)); |
816 | | // CDEF needs the top buffer to be saved by lr_copy_lpf of the |
817 | | // previous sbrow |
818 | 151k | if (sby) { |
819 | 132k | int prog = atomic_load(&f->frame_thread.copy_lpf_progress[(sby - 1) >> 5]); |
820 | 132k | if (~prog & (1U << ((sby - 1) & 31))) { |
821 | 14.6k | t->type = DAV1D_TASK_TYPE_CDEF; |
822 | 14.6k | t->recon_progress = t->deblock_progress = 0; |
823 | 14.6k | add_pending(f, t); |
824 | 14.6k | pthread_mutex_lock(&ttd->lock); |
825 | 14.6k | continue; |
826 | 14.6k | } |
827 | 132k | } |
828 | 151k | } |
829 | | // fall-through |
830 | 633k | case DAV1D_TASK_TYPE_CDEF: |
831 | 633k | if (f->seq_hdr->cdef) { |
832 | 393k | if (!atomic_load(&f->task_thread.error)) |
833 | 250k | f->bd_fn.filter_sbrow_cdef(tc, sby); |
834 | 393k | reset_task_cur_async(ttd, t->frame_idx, c->n_fc); |
835 | 393k | if (!atomic_fetch_or(&ttd->cond_signaled, 1)) |
836 | 105k | pthread_cond_signal(&ttd->cond); |
837 | 393k | } |
838 | | // fall-through |
839 | 639k | case DAV1D_TASK_TYPE_SUPER_RESOLUTION: |
840 | 639k | if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) |
841 | 82.0k | if (!atomic_load(&f->task_thread.error)) |
842 | 66.5k | f->bd_fn.filter_sbrow_resize(f, sby); |
843 | | // fall-through |
844 | 639k | case DAV1D_TASK_TYPE_LOOP_RESTORATION: |
845 | 639k | if (!atomic_load(&f->task_thread.error) && f->lf.restore_planes) |
846 | 249k | f->bd_fn.filter_sbrow_lr(f, sby); |
847 | | // fall-through |
848 | 725k | case DAV1D_TASK_TYPE_RECONSTRUCTION_PROGRESS: |
849 | | // dummy to cover for no post-filters |
850 | 1.45M | case DAV1D_TASK_TYPE_ENTROPY_PROGRESS: |
851 | | // dummy to convert tile progress to frame |
852 | 1.45M | break; |
853 | 0 | default: abort(); |
854 | 3.50M | } |
855 | | // if task completed [typically LR], signal picture progress as per below |
856 | 1.45M | const int uses_2pass = c->n_fc > 1; |
857 | 1.45M | const int sbh = f->sbh; |
858 | 1.45M | const int sbsz = f->sb_step * 4; |
859 | 1.45M | if (t->type == DAV1D_TASK_TYPE_ENTROPY_PROGRESS) { |
860 | 732k | error = atomic_load(&f->task_thread.error); |
861 | 732k | const unsigned y = sby + 1 == sbh ? UINT_MAX : (unsigned)(sby + 1) * sbsz; |
862 | 732k | assert(c->n_fc > 1); |
863 | 732k | if (f->sr_cur.p.data[0] /* upon flush, this can be free'ed already */) |
864 | 732k | atomic_store(&f->sr_cur.progress[0], error ? FRAME_ERROR : y); |
865 | 732k | atomic_store(&f->frame_thread.entropy_progress, |
866 | 732k | error ? TILE_ERROR : sby + 1); |
867 | 732k | if (sby + 1 == sbh) |
868 | 732k | atomic_store(&f->task_thread.done[1], 1); |
869 | 732k | pthread_mutex_lock(&ttd->lock); |
870 | 732k | const int num_tasks = atomic_fetch_sub(&f->task_thread.task_counter, 1) - 1; |
871 | 732k | if (sby + 1 < sbh && num_tasks) { |
872 | 477k | reset_task_cur(c, ttd, t->frame_idx); |
873 | 477k | continue; |
874 | 477k | } |
875 | 255k | if (!num_tasks && atomic_load(&f->task_thread.done[0]) && |
876 | 255k | atomic_load(&f->task_thread.done[1])) |
877 | 31.2k | { |
878 | 31.2k | error = atomic_load(&f->task_thread.error); |
879 | 31.2k | dav1d_decode_frame_exit(f, error == 1 ? DAV1D_ERR(EINVAL) : |
880 | 31.2k | error ? DAV1D_ERR(ENOMEM) : 0); |
881 | 31.2k | f->n_tile_data = 0; |
882 | 31.2k | pthread_cond_signal(&f->task_thread.cond); |
883 | 31.2k | } |
884 | 255k | reset_task_cur(c, ttd, t->frame_idx); |
885 | 255k | continue; |
886 | 732k | } |
887 | | // t->type != DAV1D_TASK_TYPE_ENTROPY_PROGRESS |
888 | 1.45M | atomic_fetch_or(&f->frame_thread.frame_progress[sby >> 5], |
889 | 722k | 1U << (sby & 31)); |
890 | 722k | pthread_mutex_lock(&f->task_thread.lock); |
891 | 722k | sby = get_frame_progress(c, f); |
892 | 722k | error = atomic_load(&f->task_thread.error); |
893 | 722k | const unsigned y = sby + 1 == sbh ? UINT_MAX : (unsigned)(sby + 1) * sbsz; |
894 | 726k | if (c->n_fc > 1 && f->sr_cur.p.data[0] /* upon flush, this can be free'ed already */) |
895 | 726k | atomic_store(&f->sr_cur.progress[1], error ? FRAME_ERROR : y); |
896 | 722k | pthread_mutex_unlock(&f->task_thread.lock); |
897 | 722k | if (sby + 1 == sbh) |
898 | 722k | atomic_store(&f->task_thread.done[0], 1); |
899 | 722k | pthread_mutex_lock(&ttd->lock); |
900 | 722k | const int num_tasks = atomic_fetch_sub(&f->task_thread.task_counter, 1) - 1; |
901 | 722k | if (sby + 1 < sbh && num_tasks) { |
902 | 237k | reset_task_cur(c, ttd, t->frame_idx); |
903 | 237k | continue; |
904 | 237k | } |
905 | 484k | if (!num_tasks && atomic_load(&f->task_thread.done[0]) && |
906 | 214k | (!uses_2pass || atomic_load(&f->task_thread.done[1]))) |
907 | 214k | { |
908 | 214k | error = atomic_load(&f->task_thread.error); |
909 | 214k | dav1d_decode_frame_exit(f, error == 1 ? DAV1D_ERR(EINVAL) : |
910 | 214k | error ? DAV1D_ERR(ENOMEM) : 0); |
911 | 214k | f->n_tile_data = 0; |
912 | 214k | pthread_cond_signal(&f->task_thread.cond); |
913 | 214k | } |
914 | 484k | reset_task_cur(c, ttd, t->frame_idx); |
915 | 484k | } |
916 | 2.73M | pthread_mutex_unlock(&ttd->lock); |
917 | | |
918 | | return NULL; |
919 | 2.74M | } |