Coverage Report

Created: 2026-09-13 06:34

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/libde265/libde265/image.cc
Line
Count
Source
1
/*
2
 * H.265 video codec.
3
 * Copyright (c) 2013-2014 struktur AG, Dirk Farin <farin@struktur.de>
4
 *
5
 * This file is part of libde265.
6
 *
7
 * libde265 is free software: you can redistribute it and/or modify
8
 * it under the terms of the GNU Lesser General Public License as
9
 * published by the Free Software Foundation, either version 3 of
10
 * the License, or (at your option) any later version.
11
 *
12
 * libde265 is distributed in the hope that it will be useful,
13
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
14
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
15
 * GNU Lesser General Public License for more details.
16
 *
17
 * You should have received a copy of the GNU Lesser General Public License
18
 * along with libde265.  If not, see <http://www.gnu.org/licenses/>.
19
 */
20
21
#include "image.h"
22
#include "decctx.h"
23
24
#include <atomic>
25
26
#include <stdlib.h>
27
#include <string.h>
28
#include <assert.h>
29
30
#include <limits>
31
32
33
#ifdef HAVE_MALLOC_H
34
#include <malloc.h>
35
#endif
36
37
#ifdef HAVE_SSE4_1
38
// SSE code processes 128bit per iteration and thus might read more data
39
// than is later actually used.
40
269k
#define MEMORY_PADDING  16
41
#else
42
#define MEMORY_PADDING  0
43
#endif
44
45
63.3k
#define STANDARD_ALIGNMENT 16
46
47
#if defined(__MINGW32__)
48
#define ALLOC_ALIGNED(alignment, size)         __mingw_aligned_malloc((size), (alignment))
49
#define FREE_ALIGNED(mem)                      __mingw_aligned_free((mem))
50
#elif defined(_MSC_VER)
51
#define ALLOC_ALIGNED(alignment, size)         _aligned_malloc((size), (alignment))
52
#define FREE_ALIGNED(mem)                      _aligned_free((mem))
53
#elif defined(HAVE_POSIX_MEMALIGN)
54
188k
static inline void *ALLOC_ALIGNED(size_t alignment, size_t size) {
55
188k
    void *mem = nullptr;
56
188k
    if (posix_memalign(&mem, alignment, size) != 0) {
57
0
        return nullptr;
58
0
    }
59
188k
    return mem;
60
188k
};
61
188k
#define FREE_ALIGNED(mem)                      free((mem))
62
#else
63
#define ALLOC_ALIGNED(alignment, size)      memalign((alignment), (size))
64
#define FREE_ALIGNED(mem)                   free((mem))
65
#endif
66
67
188k
#define ALLOC_ALIGNED_16(size)              ALLOC_ALIGNED(16, size)
68
69
LIBDE265_API void* de265_alloc_image_plane(struct de265_image* img, int cIdx,
70
                                           void* inputdata, int inputstride, void *userdata)
71
0
{
72
0
  int alignment = STANDARD_ALIGNMENT;
73
0
  uint32_t stride = (img->get_width(cIdx) + alignment-1) / alignment * alignment;
74
0
  uint32_t height = img->get_height(cIdx);
75
76
  // size computed in size_t: stride*height can exceed UINT32_MAX for large planes
77
0
  uint8_t* p = static_cast<uint8_t*>(ALLOC_ALIGNED_16(static_cast<size_t>(stride) * height + MEMORY_PADDING));
78
79
0
  if (p==nullptr) { return nullptr; }
80
81
0
  img->set_image_plane(cIdx, p, stride, userdata);
82
83
  // copy input data if provided
84
85
0
  if (inputdata != nullptr) {
86
0
    if (inputstride == static_cast<int>(stride)) {
87
0
      memcpy(p, inputdata, static_cast<size_t>(stride) * height);
88
0
    }
89
0
    else {
90
0
      for (uint32_t y=0;y<height;y++) {
91
0
        memcpy(p + static_cast<size_t>(y) * stride,
92
0
               static_cast<char*>(inputdata) + static_cast<size_t>(inputstride) * y,
93
0
               inputstride);
94
0
      }
95
0
    }
96
0
  }
97
98
0
  return p;
99
0
}
100
101
102
LIBDE265_API void de265_free_image_plane(struct de265_image* img, int cIdx)
103
0
{
104
0
  uint8_t* p = img->get_image_plane(cIdx);
105
0
  assert(p);
106
0
  FREE_ALIGNED(p);
107
0
}
108
109
110
static int  de265_image_get_buffer(de265_decoder_context* ctx,
111
                                   de265_image_spec* spec, de265_image* img, void* userdata)
112
63.3k
{
113
63.3k
  const uint32_t rawChromaWidth  = spec->width  / img->SubWidthC;
114
63.3k
  const uint32_t rawChromaHeight = spec->height / img->SubHeightC;
115
116
63.3k
  uint32_t luma_stride   = (spec->width    + spec->alignment-1) / spec->alignment * spec->alignment;
117
63.3k
  uint32_t chroma_stride = (rawChromaWidth + spec->alignment-1) / spec->alignment * spec->alignment;
118
119
63.3k
  assert(img->BitDepth_Y >= 8 && img->BitDepth_Y <= 16);
120
63.3k
  assert(img->BitDepth_C >= 8 && img->BitDepth_C <= 16);
121
122
63.3k
  uint32_t luma_bpl   = luma_stride   * ((img->BitDepth_Y+7)/8);
123
63.3k
  uint32_t chroma_bpl = chroma_stride * ((img->BitDepth_C+7)/8);
124
125
63.3k
  uint32_t luma_height   = spec->height;
126
63.3k
  uint32_t chroma_height = rawChromaHeight;
127
128
63.3k
  bool alloc_failed = false;
129
130
  // Compute the plane sizes in size_t. Each operand fits in uint32_t, but the
131
  // height * bytes-per-line product can exceed UINT32_MAX for large frames, so
132
  // the multiplication must be done in 64 bits. Computing it in 32 bits wraps
133
  // the allocation size to a small value while fill_image() later writes the
134
  // real (size_t) size -> heap buffer overflow (GHSA-vv8h-932h-7r86).
135
63.3k
  uint8_t* p[3] = { nullptr,nullptr,nullptr };
136
63.3k
  p[0] = static_cast<uint8_t*>(ALLOC_ALIGNED_16(static_cast<size_t>(luma_height) * luma_bpl + MEMORY_PADDING));
137
63.3k
  if (p[0]==nullptr) { alloc_failed=true; }
138
139
63.3k
  if (img->get_chroma_format() != de265_chroma_mono) {
140
62.7k
    p[1] = static_cast<uint8_t*>(ALLOC_ALIGNED_16(static_cast<size_t>(chroma_height) * chroma_bpl + MEMORY_PADDING));
141
62.7k
    p[2] = static_cast<uint8_t*>(ALLOC_ALIGNED_16(static_cast<size_t>(chroma_height) * chroma_bpl + MEMORY_PADDING));
142
143
62.7k
    if (p[1]==nullptr || p[2]==nullptr) { alloc_failed=true; }
144
62.7k
  }
145
628
  else {
146
628
    p[1] = nullptr;
147
628
    p[2] = nullptr;
148
628
    chroma_stride = 0;
149
628
  }
150
151
63.3k
  if (alloc_failed) {
152
0
    for (int i=0;i<3;i++)
153
0
      if (p[i]) {
154
0
        FREE_ALIGNED(p[i]);
155
0
      }
156
157
0
    return 0;
158
0
  }
159
160
63.3k
  img->set_image_plane(0, p[0], luma_stride, nullptr);
161
63.3k
  img->set_image_plane(1, p[1], chroma_stride, nullptr);
162
63.3k
  img->set_image_plane(2, p[2], chroma_stride, nullptr);
163
164
63.3k
  img->fill_image(0,0,0);
165
166
63.3k
  return 1;
167
63.3k
}
168
169
static void de265_image_release_buffer(de265_decoder_context* ctx,
170
                                       de265_image* img, void* userdata)
171
63.3k
{
172
253k
  for (int i=0;i<3;i++) {
173
190k
    uint8_t* p = img->get_image_plane(i);
174
190k
    if (p) {
175
188k
      FREE_ALIGNED(p);
176
188k
    }
177
190k
  }
178
63.3k
}
179
180
181
de265_image_allocation de265_image::default_image_allocation = {
182
  de265_image_get_buffer,
183
  de265_image_release_buffer
184
};
185
186
187
void de265_image::set_image_plane(int cIdx, uint8_t* mem, ptrdiff_t stride, void *userdata)
188
190k
{
189
190k
  pixels[cIdx] = mem;
190
190k
  plane_user_data[cIdx] = userdata;
191
192
190k
  if (cIdx==0) { this->stride        = stride; }
193
126k
  else         { this->chroma_stride = stride; }
194
190k
}
195
196
197
68.8k
de265_image::de265_image() = default;
198
199
200
de265_error de265_image::alloc_image(int w,int h, enum de265_chroma c,
201
                                     std::shared_ptr<const seq_parameter_set> sps, bool allocMetadata,
202
                                     decoder_context* dctx,
203
                                     //encoder_context* ectx,
204
                                     de265_PTS pts, void* user_data,
205
                                     bool useCustomAllocFunc)
206
63.3k
{
207
  //if (allocMetadata) { assert(sps); }
208
63.3k
  if (allocMetadata) { assert(sps); }
209
210
63.3k
  if (sps) { this->sps = sps; }
211
212
63.3k
  release(); /* TODO: review code for efficient allocation when arrays are already
213
                allocated to the requested size. Without the release, the old image-data
214
                will not be freed. */
215
216
63.3k
  static std::atomic<uint32_t> s_next_image_ID(0);
217
63.3k
  ID = s_next_image_ID++;
218
63.3k
  removed_at_picture_id = std::numeric_limits<uint32_t>::max();
219
220
63.3k
  decctx = dctx;
221
  //encctx = ectx;
222
223
  // --- allocate image buffer ---
224
225
63.3k
  chroma_format= c;
226
227
63.3k
  width = w;
228
63.3k
  height = h;
229
63.3k
  chroma_width = w;
230
63.3k
  chroma_height= h;
231
232
63.3k
  this->user_data = user_data;
233
63.3k
  this->pts = pts;
234
235
63.3k
  de265_image_spec spec;
236
237
63.3k
  uint8_t WinUnitX, WinUnitY;
238
239
63.3k
  switch (chroma_format) {
240
630
    case de265_chroma_mono: WinUnitX=1; WinUnitY=1; break;
241
47.9k
    case de265_chroma_420:  WinUnitX=2; WinUnitY=2; break;
242
9.89k
    case de265_chroma_422:  WinUnitX=2; WinUnitY=1; break;
243
4.92k
    case de265_chroma_444:  WinUnitX=1; WinUnitY=1; break;
244
0
    default:
245
0
      assert(0);
246
0
      WinUnitX = WinUnitY = 0;
247
63.3k
  }
248
249
63.3k
  switch (chroma_format) {
250
47.9k
  case de265_chroma_420:
251
47.9k
    spec.format = de265_image_format_YUV420P8;
252
47.9k
    chroma_width  = (chroma_width +1)/2;
253
47.9k
    chroma_height = (chroma_height+1)/2;
254
47.9k
    SubWidthC  = 2;
255
47.9k
    SubHeightC = 2;
256
47.9k
    break;
257
258
9.89k
  case de265_chroma_422:
259
9.89k
    spec.format = de265_image_format_YUV422P8;
260
9.89k
    chroma_width = (chroma_width+1)/2;
261
9.89k
    SubWidthC  = 2;
262
9.89k
    SubHeightC = 1;
263
9.89k
    break;
264
265
4.92k
  case de265_chroma_444:
266
4.92k
    spec.format = de265_image_format_YUV444P8;
267
4.92k
    SubWidthC  = 1;
268
4.92k
    SubHeightC = 1;
269
4.92k
    break;
270
271
630
  case de265_chroma_mono:
272
630
    spec.format = de265_image_format_mono8;
273
630
    chroma_width = 0;
274
630
    chroma_height= 0;
275
630
    SubWidthC  = 1;
276
630
    SubHeightC = 1;
277
630
    break;
278
279
0
  default:
280
0
    assert(false);
281
0
    break;
282
63.3k
  }
283
284
63.3k
  if (chroma_format != de265_chroma_mono && sps) {
285
62.7k
    assert(sps->SubWidthC  == SubWidthC);
286
62.7k
    assert(sps->SubHeightC == SubHeightC);
287
62.7k
  }
288
289
63.3k
  spec.width  = w;
290
63.3k
  spec.height = h;
291
63.3k
  spec.alignment = STANDARD_ALIGNMENT;
292
293
294
  // conformance window cropping
295
296
63.3k
  int left   = sps ? sps->conf_win_left_offset : 0;
297
63.3k
  int right  = sps ? sps->conf_win_right_offset : 0;
298
63.3k
  int top    = sps ? sps->conf_win_top_offset : 0;
299
63.3k
  int bottom = sps ? sps->conf_win_bottom_offset : 0;
300
301
63.3k
  if ((left+right)*WinUnitX >= width) {
302
0
    return DE265_ERROR_CODED_PARAMETER_OUT_OF_RANGE;
303
0
  }
304
305
63.3k
  if ((top+bottom)*WinUnitY >= height) {
306
3
    return DE265_ERROR_CODED_PARAMETER_OUT_OF_RANGE;
307
3
  }
308
309
63.3k
  width_confwin = width - (left+right)*WinUnitX;
310
63.3k
  height_confwin= height- (top+bottom)*WinUnitY;
311
63.3k
  chroma_width_confwin = chroma_width -left-right;
312
63.3k
  chroma_height_confwin= chroma_height-top-bottom;
313
314
63.3k
  spec.crop_left  = left *WinUnitX;
315
63.3k
  spec.crop_right = right*WinUnitX;
316
63.3k
  spec.crop_top   = top   *WinUnitY;
317
63.3k
  spec.crop_bottom= bottom*WinUnitY;
318
319
63.3k
  spec.visible_width = width_confwin;
320
63.3k
  spec.visible_height= height_confwin;
321
322
323
63.3k
  BitDepth_Y = (sps==nullptr) ? 8 : sps->BitDepth_Y;
324
63.3k
  BitDepth_C = (sps==nullptr) ? 8 : sps->BitDepth_C;
325
326
63.3k
  bpp_shift[0] = (BitDepth_Y <= 8) ? 0 : 1;
327
63.3k
  bpp_shift[1] = (BitDepth_C <= 8) ? 0 : 1;
328
63.3k
  bpp_shift[2] = bpp_shift[1];
329
330
331
  // allocate memory and set conformance window pointers
332
333
63.3k
  void* alloc_userdata = nullptr;
334
63.3k
  if (decctx) alloc_userdata = decctx->param_image_allocation_userdata;
335
  // if (encctx) alloc_userdata = encctx->param_image_allocation_userdata; // actually not needed
336
337
  /*
338
  if (encctx && useCustomAllocFunc) {
339
    encoder_image_release_func = encctx->release_func;
340
341
    // if we do not provide a release function, use our own
342
343
    if (encoder_image_release_func == nullptr) {
344
      image_allocation_functions = de265_image::default_image_allocation;
345
    }
346
    else {
347
      image_allocation_functions.get_buffer     = nullptr;
348
      image_allocation_functions.release_buffer = nullptr;
349
    }
350
  }
351
63.3k
  else*/ if (decctx && useCustomAllocFunc) {
352
20.7k
    image_allocation_functions = decctx->param_image_allocation_functions;
353
20.7k
  }
354
42.6k
  else {
355
42.6k
    image_allocation_functions = de265_image::default_image_allocation;
356
42.6k
  }
357
358
63.3k
  bool mem_alloc_success = true;
359
360
63.3k
  if (image_allocation_functions.get_buffer != nullptr) {
361
63.3k
    mem_alloc_success = image_allocation_functions.get_buffer(decctx, &spec, this,
362
63.3k
                                                              alloc_userdata);
363
364
63.3k
    pixels_confwin[0] = pixels[0] + left*WinUnitX + top*WinUnitY*stride;
365
366
63.3k
    if (chroma_format != de265_chroma_mono) {
367
62.7k
      pixels_confwin[1] = pixels[1] + left + top*chroma_stride;
368
62.7k
      pixels_confwin[2] = pixels[2] + left + top*chroma_stride;
369
62.7k
    }
370
626
    else {
371
626
      pixels_confwin[1] = nullptr;
372
626
      pixels_confwin[2] = nullptr;
373
626
    }
374
375
    // check for memory shortage
376
377
63.3k
    if (!mem_alloc_success)
378
0
      {
379
0
        return DE265_ERROR_OUT_OF_MEMORY;
380
0
      }
381
63.3k
  }
382
383
  //alloc_functions = *allocfunc;
384
  //alloc_userdata  = userdata;
385
386
  // --- allocate decoding info arrays ---
387
388
63.3k
  if (allocMetadata) {
389
    // intra pred mode
390
391
48.1k
    mem_alloc_success &= intraPredMode.alloc(sps->PicWidthInMinPUs, sps->PicHeightInMinPUs,
392
48.1k
                                             sps->Log2MinPUSize);
393
394
48.1k
    mem_alloc_success &= intraPredModeC.alloc(sps->PicWidthInMinPUs, sps->PicHeightInMinPUs,
395
48.1k
                                              sps->Log2MinPUSize);
396
397
    // cb info
398
399
48.1k
    mem_alloc_success &= cb_info.alloc(sps->PicWidthInMinCbsY, sps->PicHeightInMinCbsY,
400
48.1k
                                       sps->Log2MinCbSizeY);
401
402
    // pb info
403
404
48.1k
    int puWidth  = sps->PicWidthInMinCbsY  << (sps->Log2MinCbSizeY -2);
405
48.1k
    int puHeight = sps->PicHeightInMinCbsY << (sps->Log2MinCbSizeY -2);
406
407
48.1k
    mem_alloc_success &= pb_info.alloc(puWidth,puHeight, 2);
408
409
410
    // tu info
411
412
48.1k
    mem_alloc_success &= tu_info.alloc(sps->PicWidthInTbsY, sps->PicHeightInTbsY,
413
48.1k
                                       sps->Log2MinTrafoSize);
414
415
    // deblk info
416
417
48.1k
    int deblk_w = (sps->pic_width_in_luma_samples +3)/4;
418
48.1k
    int deblk_h = (sps->pic_height_in_luma_samples+3)/4;
419
420
48.1k
    mem_alloc_success &= deblk_info.alloc(deblk_w, deblk_h, 2);
421
422
    // CTB info
423
424
48.1k
    if (ctb_info.width_in_units  != sps->PicWidthInCtbsY  ||
425
0
        ctb_info.height_in_units != sps->PicHeightInCtbsY ||
426
0
        ctb_info.log2unitSize    != sps->Log2CtbSizeY)
427
48.1k
      {
428
48.1k
        delete[] ctb_progress;
429
430
48.1k
        mem_alloc_success &= ctb_info.alloc(sps->PicWidthInCtbsY, sps->PicHeightInCtbsY,
431
48.1k
                                            sps->Log2CtbSizeY);
432
433
48.1k
        ctb_progress = new de265_progress_lock[ ctb_info.data_size ];
434
48.1k
      }
435
436
437
    // check for memory shortage
438
439
48.1k
    if (!mem_alloc_success)
440
0
      {
441
0
        return DE265_ERROR_OUT_OF_MEMORY;
442
0
      }
443
48.1k
  }
444
445
63.3k
  return DE265_OK;
446
63.3k
}
447
448
449
de265_image::~de265_image()
450
68.8k
{
451
68.8k
  release();
452
453
  // free progress locks
454
455
68.8k
  if (ctb_progress) {
456
48.1k
    delete[] ctb_progress;
457
48.1k
  }
458
68.8k
}
459
460
461
void de265_image::release()
462
132k
{
463
  // free image memory
464
465
132k
  if (pixels[0])
466
63.3k
    {
467
      /*
468
      if (encoder_image_release_func != nullptr) {
469
        encoder_image_release_func(encctx, this,
470
                                   encctx->param_image_allocation_userdata);
471
      }
472
63.3k
      else*/ {
473
63.3k
        image_allocation_functions.release_buffer(decctx, this,
474
63.3k
                                                  decctx ?
475
63.3k
                                                  decctx->param_image_allocation_userdata :
476
63.3k
                                                  nullptr);
477
63.3k
      }
478
479
253k
      for (int i=0;i<3;i++)
480
190k
        {
481
190k
          pixels[i] = nullptr;
482
190k
          pixels_confwin[i] = nullptr;
483
190k
        }
484
63.3k
    }
485
486
  // free slices
487
488
153k
  for (size_t i=0;i<slices.size();i++) {
489
21.2k
    delete slices[i];
490
21.2k
  }
491
132k
  slices.clear();
492
132k
}
493
494
495
void de265_image::fill_plane(int channel, int value)
496
269k
{
497
269k
  int bytes_per_pixel = get_bytes_per_pixel(channel);
498
269k
  assert(value >= 0); // needed for the shift operation in the check below
499
500
  // Each plane is allocated with MEMORY_PADDING trailing bytes for safe SSE overread; the
501
  // memsets below cover that padding too so it never contains uninitialized heap data.
502
269k
  const size_t plane_bytes =
503
269k
      (channel == 0 ? static_cast<size_t>(stride) * height
504
269k
                    : static_cast<size_t>(chroma_stride) * chroma_height)
505
269k
      * bytes_per_pixel;
506
507
269k
  if (bytes_per_pixel == 1) {
508
187k
    memset(pixels[channel], value, plane_bytes + MEMORY_PADDING);
509
187k
  }
510
82.3k
  else if ((value >> 8) == (value & 0xFF)) {
511
59.0k
    assert(bytes_per_pixel == 2);
512
513
    // if we fill the same byte value to all bytes, we can still use memset()
514
59.0k
    memset(pixels[channel], 0, plane_bytes + MEMORY_PADDING);
515
59.0k
  }
516
23.3k
  else {
517
23.3k
    assert(bytes_per_pixel == 2);
518
23.3k
    uint16_t v = value;
519
520
23.3k
    if (channel==0) {
521
      // copy value into first row
522
7.83M
      for (int x = 0; x < width; x++) {
523
7.83M
        *reinterpret_cast<uint16_t*>(&pixels[channel][2 * x]) = v;
524
7.83M
      }
525
526
      // copy first row into remaining rows
527
649k
      for (int y = 1; y < height; y++) {
528
641k
        memcpy(pixels[channel] + y * stride * 2, pixels[channel], chroma_width * 2);
529
641k
      }
530
8.40k
    }
531
14.8k
    else {
532
      // copy value into first row
533
10.2M
      for (int x = 0; x < chroma_width; x++) {
534
10.1M
        *reinterpret_cast<uint16_t*>(&pixels[channel][2 * x]) = v;
535
10.1M
      }
536
537
      // copy first row into remaining rows
538
798k
      for (int y = 1; y < chroma_height; y++) {
539
783k
        memcpy(pixels[channel] + y * chroma_stride * 2, pixels[channel], chroma_width * 2);
540
783k
      }
541
14.8k
    }
542
543
23.3k
#if MEMORY_PADDING > 0
544
23.3k
    memset(pixels[channel] + plane_bytes, 0, MEMORY_PADDING);
545
23.3k
#endif
546
23.3k
  }
547
269k
}
548
549
550
void de265_image::fill_image(int y,int cb,int cr)
551
90.5k
{
552
90.5k
  if (pixels[0]) {
553
90.5k
    fill_plane(0, y);
554
90.5k
  }
555
556
90.5k
  if (pixels[1]) {
557
89.6k
    fill_plane(1, cb);
558
89.6k
  }
559
560
90.5k
  if (pixels[2]) {
561
89.6k
    fill_plane(2, cr);
562
89.6k
  }
563
90.5k
}
564
565
566
de265_error de265_image::copy_image(const de265_image* src)
567
0
{
568
  /* TODO: actually, since we allocate the image only for internal purpose, we
569
     do not have to call the external allocation routines for this. However, then
570
     we have to track for each image how to release it again.
571
     Another option would be to safe the copied data not in an de265_image at all.
572
  */
573
574
0
  de265_error err = alloc_image(src->width, src->height, src->chroma_format, src->sps, false,
575
0
                                src->decctx, /*src->encctx,*/ src->pts, src->user_data, false);
576
0
  if (err != DE265_OK) {
577
0
    return err;
578
0
  }
579
580
0
  copy_lines_from(src, 0, src->height);
581
582
0
  return err;
583
0
}
584
585
586
// end = last line + 1
587
void de265_image::copy_lines_from(const de265_image* src, int first, int end)
588
64.8k
{
589
64.8k
  if (end > src->height) end=src->height;
590
591
64.8k
  assert(first % 2 == 0);
592
64.8k
  assert(end   % 2 == 0);
593
594
64.8k
  int luma_bpp   = (sps->BitDepth_Y+7)/8;
595
64.8k
  int chroma_bpp = (sps->BitDepth_C+7)/8;
596
597
64.8k
  if (src->stride == stride) {
598
64.8k
    memcpy(pixels[0]      + first*stride * luma_bpp,
599
64.8k
           src->pixels[0] + first*src->stride * luma_bpp,
600
64.8k
           (end-first)*stride * luma_bpp);
601
64.8k
  }
602
0
  else {
603
0
    for (int yp=first;yp<end;yp++) {
604
0
      memcpy(pixels[0]+yp*stride * luma_bpp,
605
0
             src->pixels[0]+yp*src->stride * luma_bpp,
606
0
             src->width * luma_bpp);
607
0
    }
608
0
  }
609
610
64.8k
  int first_chroma = first / src->SubHeightC;
611
64.8k
  int end_chroma   = end   / src->SubHeightC;
612
613
64.8k
  if (src->chroma_format != de265_chroma_mono) {
614
64.5k
    if (src->chroma_stride == chroma_stride) {
615
64.5k
      memcpy(pixels[1]      + first_chroma*chroma_stride * chroma_bpp,
616
64.5k
             src->pixels[1] + first_chroma*chroma_stride * chroma_bpp,
617
64.5k
             (end_chroma-first_chroma) * chroma_stride * chroma_bpp);
618
64.5k
      memcpy(pixels[2]      + first_chroma*chroma_stride * chroma_bpp,
619
64.5k
             src->pixels[2] + first_chroma*chroma_stride * chroma_bpp,
620
64.5k
             (end_chroma-first_chroma) * chroma_stride * chroma_bpp);
621
64.5k
    }
622
0
    else {
623
0
      for (int y=first_chroma;y<end_chroma;y++) {
624
0
        memcpy(pixels[1]+y*chroma_stride * chroma_bpp,
625
0
               src->pixels[1]+y*src->chroma_stride * chroma_bpp,
626
0
               src->chroma_width * chroma_bpp);
627
0
        memcpy(pixels[2]+y*chroma_stride * chroma_bpp,
628
0
               src->pixels[2]+y*src->chroma_stride * chroma_bpp,
629
0
               src->chroma_width * chroma_bpp);
630
0
      }
631
0
    }
632
64.5k
  }
633
64.8k
}
634
635
636
void de265_image::exchange_pixel_data_with(de265_image& b)
637
15.1k
{
638
60.7k
  for (int i=0;i<3;i++) {
639
45.5k
    std::swap(pixels[i], b.pixels[i]);
640
45.5k
    std::swap(pixels_confwin[i], b.pixels_confwin[i]);
641
45.5k
    std::swap(plane_user_data[i], b.plane_user_data[i]);
642
45.5k
  }
643
644
15.1k
  std::swap(stride, b.stride);
645
15.1k
  std::swap(chroma_stride, b.chroma_stride);
646
15.1k
  std::swap(image_allocation_functions, b.image_allocation_functions);
647
15.1k
}
648
649
650
void de265_image::thread_start(int nThreads)
651
61.5k
{
652
61.5k
  std::unique_lock<std::mutex> lock(mutex);
653
654
  //printf("nThreads before: %d %d\n",nThreadsQueued, nThreadsTotal);
655
656
61.5k
  nThreadsQueued += nThreads;
657
61.5k
  nThreadsTotal += nThreads;
658
659
  //printf("nThreads after: %d %d\n",nThreadsQueued, nThreadsTotal);
660
61.5k
}
661
662
void de265_image::thread_run(const thread_task* task)
663
268k
{
664
268k
  std::unique_lock<std::mutex> lock(mutex);
665
666
  //printf("run thread %s\n", task->name().c_str());
667
668
268k
  nThreadsQueued--;
669
268k
  nThreadsRunning++;
670
268k
}
671
672
void de265_image::thread_blocks()
673
0
{
674
0
  std::unique_lock<std::mutex> lock(mutex);
675
676
0
  nThreadsRunning--;
677
0
  nThreadsBlocked++;
678
0
}
679
680
void de265_image::thread_unblocks()
681
0
{
682
0
  std::unique_lock<std::mutex> lock(mutex);
683
684
0
  nThreadsBlocked--;
685
0
  nThreadsRunning++;
686
0
}
687
688
void de265_image::thread_finishes(const thread_task* task)
689
268k
{
690
  //printf("finish thread %s\n", task->name().c_str());
691
692
268k
  std::unique_lock<std::mutex> lock(mutex);
693
694
268k
  nThreadsRunning--;
695
268k
  nThreadsFinished++;
696
268k
  assert(nThreadsRunning >= 0);
697
698
268k
  if (nThreadsFinished==nThreadsTotal) {
699
25.3k
    finished_cond.notify_all();
700
25.3k
  }
701
268k
}
702
703
void de265_image::wait_for_progress(thread_task* task, int ctbx,int ctby, int progress)
704
854k
{
705
854k
  const int ctbW = sps->PicWidthInCtbsY;
706
707
854k
  wait_for_progress(task, ctbx + ctbW*ctby, progress);
708
854k
}
709
710
void de265_image::wait_for_progress(thread_task* task, int ctbAddrRS, int progress)
711
854k
{
712
854k
  if (task==nullptr) { return; }
713
714
854k
  de265_progress_lock* progresslock = &ctb_progress[ctbAddrRS];
715
854k
  if (progresslock->get_progress() < progress) {
716
0
    thread_blocks();
717
718
0
    assert(task!=nullptr);
719
0
    task->state = thread_task::Blocked;
720
721
    /* TODO: check whether we are the first blocked task in the list.
722
       If we are, we have to conceal input errors.
723
       Simplest concealment: do not block.
724
    */
725
726
0
    progresslock->wait_for_progress(progress);
727
0
    task->state = thread_task::Running;
728
0
    thread_unblocks();
729
0
  }
730
854k
}
731
732
733
void de265_image::wait_for_completion()
734
40.6k
{
735
40.6k
  std::unique_lock<std::mutex> lock(mutex);
736
737
65.9k
  while (nThreadsFinished!=nThreadsTotal) {
738
25.3k
    finished_cond.wait(lock);
739
25.3k
  }
740
40.6k
}
741
742
bool de265_image::debug_is_completed() const
743
0
{
744
0
  return nThreadsFinished==nThreadsTotal;
745
0
}
746
747
748
749
void de265_image::clear_metadata()
750
20.9k
{
751
  // TODO: maybe we could avoid the memset by ensuring that all data is written to
752
  // during decoding (especially log2CbSize), but it is unlikely to be faster than the memset.
753
754
20.9k
  cb_info.clear();
755
20.9k
  intraPredMode.clear();
756
  //tu_info.clear();  // done on the fly
757
20.9k
  ctb_info.clear();
758
20.9k
  deblk_info.clear();
759
760
  // --- reset CTB progresses ---
761
762
3.26M
  for (int i=0;i<ctb_info.data_size;i++) {
763
3.24M
    ctb_progress[i].reset(CTB_PROGRESS_NONE);
764
3.24M
  }
765
20.9k
}
766
767
768
void de265_image::set_mv_info(int x,int y, int nPbW,int nPbH, const PBMotion& mv)
769
7.84M
{
770
7.84M
  int log2PuSize = 2;
771
772
7.84M
  int xPu = x >> log2PuSize;
773
7.84M
  int yPu = y >> log2PuSize;
774
7.84M
  int wPu = nPbW >> log2PuSize;
775
7.84M
  int hPu = nPbH >> log2PuSize;
776
777
7.84M
  int stride = pb_info.width_in_units;
778
779
26.1M
  for (int pby=0;pby<hPu;pby++)
780
73.8M
    for (int pbx=0;pbx<wPu;pbx++)
781
55.5M
      {
782
55.5M
        pb_info[ xPu+pbx + (yPu+pby)*stride ] = mv;
783
55.5M
      }
784
7.84M
}
785
786
787
bool de265_image::available_zscan(int xCurr,int yCurr, int xN,int yN) const
788
60.5M
{
789
60.5M
  if (xN<0 || yN<0) return false;
790
54.3M
  if (xN>=sps->pic_width_in_luma_samples ||
791
54.3M
      yN>=sps->pic_height_in_luma_samples) return false;
792
793
53.9M
  int minBlockAddrN = pps->scan->MinTbAddrZS[ (xN>>sps->Log2MinTrafoSize) +
794
53.9M
                                        (yN>>sps->Log2MinTrafoSize) * sps->PicWidthInTbsY ];
795
53.9M
  int minBlockAddrCurr = pps->scan->MinTbAddrZS[ (xCurr>>sps->Log2MinTrafoSize) +
796
53.9M
                                           (yCurr>>sps->Log2MinTrafoSize) * sps->PicWidthInTbsY ];
797
798
53.9M
  if (minBlockAddrN > minBlockAddrCurr) return false;
799
800
50.9M
  int xCurrCtb = xCurr >> sps->Log2CtbSizeY;
801
50.9M
  int yCurrCtb = yCurr >> sps->Log2CtbSizeY;
802
50.9M
  int xNCtb = xN >> sps->Log2CtbSizeY;
803
50.9M
  int yNCtb = yN >> sps->Log2CtbSizeY;
804
805
50.9M
  if (get_SliceAddrRS(xCurrCtb,yCurrCtb) !=
806
50.9M
      get_SliceAddrRS(xNCtb,   yNCtb)) {
807
13.1k
    return false;
808
13.1k
  }
809
810
50.9M
  if (pps->scan->TileIdRS[xCurrCtb + yCurrCtb*sps->PicWidthInCtbsY] !=
811
50.9M
      pps->scan->TileIdRS[xNCtb    + yNCtb   *sps->PicWidthInCtbsY]) {
812
79.6k
    return false;
813
79.6k
  }
814
815
50.8M
  return true;
816
50.9M
}
817
818
819
bool de265_image::available_pred_blk(int xC,int yC, int nCbS, int xP, int yP,
820
                                     int nPbW, int nPbH, int partIdx, int xN,int yN) const
821
21.7M
{
822
21.7M
  logtrace(LogMotion,"C:%d;%d P:%d;%d N:%d;%d size=%d;%d\n",xC,yC,xP,yP,xN,yN,nPbW,nPbH);
823
824
21.7M
  int sameCb = (xC <= xN && xN < xC+nCbS &&
825
5.98M
                yC <= yN && yN < yC+nCbS);
826
827
21.7M
  bool availableN;
828
829
21.7M
  if (!sameCb) {
830
20.7M
    availableN = available_zscan(xP,yP,xN,yN);
831
20.7M
  }
832
967k
  else {
833
967k
    availableN = !(nPbW<<1 == nCbS && nPbH<<1 == nCbS &&  // NxN
834
12.2k
                   partIdx==1 &&
835
4.41k
                   yN >= yC+nPbH && xN < xC+nPbW);  // xN/yN inside partIdx 2
836
967k
  }
837
838
21.7M
  if (availableN && get_pred_mode(xN,yN) == MODE_INTRA) {
839
290k
    availableN = false;
840
290k
  }
841
842
21.7M
  return availableN;
843
21.7M
}