obu.c:clz:
  186|  7.45k|static inline int clz(const unsigned int mask) {
  187|  7.45k|    return __builtin_clz(mask);
  188|  7.45k|}
thread_task.c:ctz:
  182|  1.79M|static inline int ctz(const unsigned int mask) {
  183|  1.79M|    return __builtin_ctz(mask);
  184|  1.79M|}
decode.c:ctz:
  182|   192k|static inline int ctz(const unsigned int mask) {
  183|   192k|    return __builtin_ctz(mask);
  184|   192k|}
decode.c:clz:
  186|  26.8M|static inline int clz(const unsigned int mask) {
  187|  26.8M|    return __builtin_clz(mask);
  188|  26.8M|}
getbits.c:clz:
  186|   178k|static inline int clz(const unsigned int mask) {
  187|   178k|    return __builtin_clz(mask);
  188|   178k|}
lf_mask.c:clz:
  186|  6.75M|static inline int clz(const unsigned int mask) {
  187|  6.75M|    return __builtin_clz(mask);
  188|  6.75M|}
warpmv.c:clz:
  186|   311k|static inline int clz(const unsigned int mask) {
  187|   311k|    return __builtin_clz(mask);
  188|   311k|}
warpmv.c:clzll:
  190|   161k|static inline int clzll(const unsigned long long mask) {
  191|   161k|    return __builtin_clzll(mask);
  192|   161k|}
looprestoration_tmpl.c:clz:
  186|  2.19M|static inline int clz(const unsigned int mask) {
  187|  2.19M|    return __builtin_clz(mask);
  188|  2.19M|}
recon_tmpl.c:clz:
  186|   103M|static inline int clz(const unsigned int mask) {
  187|   103M|    return __builtin_clz(mask);
  188|   103M|}
cdef_apply_tmpl.c:clz:
  186|  2.44M|static inline int clz(const unsigned int mask) {
  187|  2.44M|    return __builtin_clz(mask);
  188|  2.44M|}
ipred_prepare_tmpl.c:clz:
  186|  16.4M|static inline int clz(const unsigned int mask) {
  187|  16.4M|    return __builtin_clz(mask);
  188|  16.4M|}

fg_apply_tmpl.c:PXSTRIDE:
   79|  53.3k|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  53.3k|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 53.3k, False: 4]
  ------------------
   81|  53.3k|    return x >> 1;
   82|  53.3k|}
itx_tmpl.c:PXSTRIDE:
   79|  2.36M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  2.36M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 2.36M, False: 18.4E]
  ------------------
   81|  2.36M|    return x >> 1;
   82|  2.36M|}
looprestoration_tmpl.c:PXSTRIDE:
   79|  5.24M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  5.24M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 5.24M, False: 18.4E]
  ------------------
   81|  5.24M|    return x >> 1;
   82|  5.24M|}
recon_tmpl.c:PXSTRIDE:
   79|  40.2M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  40.2M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 40.3M, False: 18.4E]
  ------------------
   81|  40.3M|    return x >> 1;
   82|  40.2M|}
cdef_apply_tmpl.c:PXSTRIDE:
   79|  78.5M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  78.5M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 78.5M, False: 53.5k]
  ------------------
   81|  78.5M|    return x >> 1;
   82|  78.5M|}
ipred_prepare_tmpl.c:PXSTRIDE:
   79|   106M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|   106M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 105M, False: 603k]
  ------------------
   81|   105M|    return x >> 1;
   82|   106M|}
ipred_prepare_tmpl.c:pixel_set:
   66|  2.14M|static inline void pixel_set(pixel *const dst, const int val, const int num) {
   67|  22.1M|    for (int n = 0; n < num; n++)
  ------------------
  |  Branch (67:21): [True: 20.0M, False: 2.14M]
  ------------------
   68|  20.0M|        dst[n] = val;
   69|  2.14M|}
lf_apply_tmpl.c:PXSTRIDE:
   79|  42.1M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  42.1M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 42.1M, False: 18.4E]
  ------------------
   81|  42.1M|    return x >> 1;
   82|  42.1M|}
lr_apply_tmpl.c:PXSTRIDE:
   79|  1.60M|static inline ptrdiff_t PXSTRIDE(const ptrdiff_t x) {
   80|  1.60M|    assert(!(x & 1));
  ------------------
  |  Branch (80:5): [True: 1.60M, False: 2]
  ------------------
   81|  1.60M|    return x >> 1;
   82|  1.60M|}

lib.c:umin:
   47|  9.41k|static inline unsigned umin(const unsigned a, const unsigned b) {
   48|  9.41k|    return a < b ? a : b;
  ------------------
  |  Branch (48:12): [True: 0, False: 9.41k]
  ------------------
   49|  9.41k|}
obu.c:ulog2:
   67|  7.45k|static inline int ulog2(const unsigned v) {
   68|  7.45k|    return 31 ^ clz(v);
   69|  7.45k|}
obu.c:imin:
   39|  1.10M|static inline int imin(const int a, const int b) {
   40|  1.10M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 983k, False: 123k]
  ------------------
   41|  1.10M|}
obu.c:imax:
   35|  1.01M|static inline int imax(const int a, const int b) {
   36|  1.01M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 118k, False: 892k]
  ------------------
   37|  1.01M|}
obu.c:iclip_u8:
   55|   182k|static inline int iclip_u8(const int v) {
   56|   182k|    return iclip(v, 0, 255);
   57|   182k|}
obu.c:iclip:
   51|   182k|static inline int iclip(const int v, const int min, const int max) {
   52|   182k|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 6.21k, False: 176k]
  |  Branch (52:28): [True: 3.04k, False: 173k]
  ------------------
   53|   182k|}
refmvs.c:imin:
   39|  71.4M|static inline int imin(const int a, const int b) {
   40|  71.4M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 30.2M, False: 41.1M]
  ------------------
   41|  71.4M|}
refmvs.c:apply_sign:
   59|  2.76M|static inline int apply_sign(const int v, const int s) {
   60|  2.76M|    return s < 0 ? -v : v;
  ------------------
  |  Branch (60:12): [True: 1.72M, False: 1.04M]
  ------------------
   61|  2.76M|}
refmvs.c:imax:
   35|  31.0M|static inline int imax(const int a, const int b) {
   36|  31.0M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 3.84M, False: 27.2M]
  ------------------
   37|  31.0M|}
refmvs.c:iclip:
   51|  30.0M|static inline int iclip(const int v, const int min, const int max) {
   52|  30.0M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 331k, False: 29.7M]
  |  Branch (52:28): [True: 394k, False: 29.3M]
  ------------------
   53|  30.0M|}
thread_task.c:imax:
   35|  10.6M|static inline int imax(const int a, const int b) {
   36|  10.6M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 1.47M, False: 9.18M]
  ------------------
   37|  10.6M|}
thread_task.c:iclip:
   51|  1.63M|static inline int iclip(const int v, const int min, const int max) {
   52|  1.63M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 626, False: 1.63M]
  |  Branch (52:28): [True: 27.9k, False: 1.60M]
  ------------------
   53|  1.63M|}
thread_task.c:umin:
   47|  13.1M|static inline unsigned umin(const unsigned a, const unsigned b) {
   48|  13.1M|    return a < b ? a : b;
  ------------------
  |  Branch (48:12): [True: 115k, False: 13.0M]
  ------------------
   49|  13.1M|}
wedge.c:imax:
   35|    256|static inline int imax(const int a, const int b) {
   36|    256|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 128, False: 128]
  ------------------
   37|    256|}
wedge.c:imin:
   39|  2.48k|static inline int imin(const int a, const int b) {
   40|  2.48k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 1.41k, False: 1.06k]
  ------------------
   41|  2.48k|}
fg_apply_tmpl.c:imin:
   39|  44.9k|static inline int imin(const int a, const int b) {
   40|  44.9k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 10.7k, False: 34.1k]
  ------------------
   41|  44.9k|}
cdf.c:imin:
   39|   118k|static inline int imin(const int a, const int b) {
   40|   118k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 29.5k, False: 88.6k]
  ------------------
   41|   118k|}
decode.c:iclip:
   51|  4.32M|static inline int iclip(const int v, const int min, const int max) {
   52|  4.32M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 207k, False: 4.11M]
  |  Branch (52:28): [True: 284k, False: 3.83M]
  ------------------
   53|  4.32M|}
decode.c:apply_sign:
   59|  1.83M|static inline int apply_sign(const int v, const int s) {
   60|  1.83M|    return s < 0 ? -v : v;
  ------------------
  |  Branch (60:12): [True: 1.37M, False: 461k]
  ------------------
   61|  1.83M|}
decode.c:apply_sign64:
   63|  1.80M|static inline int apply_sign64(const int v, const int64_t s) {
   64|  1.80M|    return s < 0 ? -v : v;
  ------------------
  |  Branch (64:12): [True: 293k, False: 1.50M]
  ------------------
   65|  1.80M|}
decode.c:ulog2:
   67|  26.8M|static inline int ulog2(const unsigned v) {
   68|  26.8M|    return 31 ^ clz(v);
   69|  26.8M|}
decode.c:imax:
   35|  33.4M|static inline int imax(const int a, const int b) {
   36|  33.4M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 13.7M, False: 19.6M]
  ------------------
   37|  33.4M|}
decode.c:imin:
   39|  76.1M|static inline int imin(const int a, const int b) {
   40|  76.1M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 40.6M, False: 35.4M]
  ------------------
   41|  76.1M|}
decode.c:iclip_u8:
   55|  3.04M|static inline int iclip_u8(const int v) {
   56|  3.04M|    return iclip(v, 0, 255);
   57|  3.04M|}
getbits.c:ulog2:
   67|   178k|static inline int ulog2(const unsigned v) {
   68|   178k|    return 31 ^ clz(v);
   69|   178k|}
getbits.c:inv_recenter:
   75|   178k|static inline unsigned inv_recenter(const unsigned r, const unsigned v) {
   76|   178k|    if (v > (r << 1))
  ------------------
  |  Branch (76:9): [True: 3.41k, False: 175k]
  ------------------
   77|  3.41k|        return v;
   78|   175k|    else if ((v & 1) == 0)
  ------------------
  |  Branch (78:14): [True: 128k, False: 46.8k]
  ------------------
   79|   128k|        return (v >> 1) + r;
   80|  46.8k|    else
   81|  46.8k|        return r - ((v + 1) >> 1);
   82|   178k|}
lf_mask.c:imin:
   39|   158M|static inline int imin(const int a, const int b) {
   40|   158M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 15.8M, False: 142M]
  ------------------
   41|   158M|}
lf_mask.c:ulog2:
   67|  6.75M|static inline int ulog2(const unsigned v) {
   68|  6.75M|    return 31 ^ clz(v);
   69|  6.75M|}
lf_mask.c:imax:
   35|  2.17M|static inline int imax(const int a, const int b) {
   36|  2.17M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 2.03M, False: 140k]
  ------------------
   37|  2.17M|}
lf_mask.c:iclip:
   51|  7.29M|static inline int iclip(const int v, const int min, const int max) {
   52|  7.29M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 728k, False: 6.57M]
  |  Branch (52:28): [True: 261k, False: 6.30M]
  ------------------
   53|  7.29M|}
msac.c:inv_recenter:
   75|   231k|static inline unsigned inv_recenter(const unsigned r, const unsigned v) {
   76|   231k|    if (v > (r << 1))
  ------------------
  |  Branch (76:9): [True: 80.6k, False: 150k]
  ------------------
   77|  80.6k|        return v;
   78|   150k|    else if ((v & 1) == 0)
  ------------------
  |  Branch (78:14): [True: 74.0k, False: 76.8k]
  ------------------
   79|  74.0k|        return (v >> 1) + r;
   80|  76.8k|    else
   81|  76.8k|        return r - ((v + 1) >> 1);
   82|   231k|}
warpmv.c:apply_sign:
   59|  1.55M|static inline int apply_sign(const int v, const int s) {
   60|  1.55M|    return s < 0 ? -v : v;
  ------------------
  |  Branch (60:12): [True: 401k, False: 1.15M]
  ------------------
   61|  1.55M|}
warpmv.c:ulog2:
   67|   311k|static inline int ulog2(const unsigned v) {
   68|   311k|    return 31 ^ clz(v);
   69|   311k|}
warpmv.c:apply_sign64:
   63|  1.42M|static inline int apply_sign64(const int v, const int64_t s) {
   64|  1.42M|    return s < 0 ? -v : v;
  ------------------
  |  Branch (64:12): [True: 247k, False: 1.17M]
  ------------------
   65|  1.42M|}
warpmv.c:iclip:
   51|  2.45M|static inline int iclip(const int v, const int min, const int max) {
   52|  2.45M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 57.2k, False: 2.39M]
  |  Branch (52:28): [True: 72.4k, False: 2.32M]
  ------------------
   53|  2.45M|}
warpmv.c:u64log2:
   71|   161k|static inline int u64log2(const uint64_t v) {
   72|   161k|    return 63 ^ clzll(v);
   73|   161k|}
itx_tmpl.c:iclip:
   51|   148M|static inline int iclip(const int v, const int min, const int max) {
   52|   148M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 10.2M, False: 138M]
  |  Branch (52:28): [True: 15.0M, False: 123M]
  ------------------
   53|   148M|}
itx_tmpl.c:imin:
   39|  61.8k|static inline int imin(const int a, const int b) {
   40|  61.8k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 8.40k, False: 53.4k]
  ------------------
   41|  61.8k|}
looprestoration_tmpl.c:iclip:
   51|   165M|static inline int iclip(const int v, const int min, const int max) {
   52|   165M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 461k, False: 165M]
  |  Branch (52:28): [True: 439k, False: 164M]
  ------------------
   53|   165M|}
looprestoration_tmpl.c:imax:
   35|   186M|static inline int imax(const int a, const int b) {
   36|   186M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 152M, False: 34.0M]
  ------------------
   37|   186M|}
looprestoration_tmpl.c:umin:
   47|   186M|static inline unsigned umin(const unsigned a, const unsigned b) {
   48|   186M|    return a < b ? a : b;
  ------------------
  |  Branch (48:12): [True: 180M, False: 6.41M]
  ------------------
   49|   186M|}
recon_tmpl.c:ulog2:
   67|   103M|static inline int ulog2(const unsigned v) {
   68|   103M|    return 31 ^ clz(v);
   69|   103M|}
recon_tmpl.c:imin:
   39|   253M|static inline int imin(const int a, const int b) {
   40|   253M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 226M, False: 27.1M]
  ------------------
   41|   253M|}
recon_tmpl.c:imax:
   35|  18.1M|static inline int imax(const int a, const int b) {
   36|  18.1M|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 12.6M, False: 5.47M]
  ------------------
   37|  18.1M|}
recon_tmpl.c:umin:
   47|   447M|static inline unsigned umin(const unsigned a, const unsigned b) {
   48|   447M|    return a < b ? a : b;
  ------------------
  |  Branch (48:12): [True: 306M, False: 141M]
  ------------------
   49|   447M|}
recon_tmpl.c:apply_sign64:
   63|  1.33M|static inline int apply_sign64(const int v, const int64_t s) {
   64|  1.33M|    return s < 0 ? -v : v;
  ------------------
  |  Branch (64:12): [True: 98.6k, False: 1.23M]
  ------------------
   65|  1.33M|}
recon_tmpl.c:iclip:
   51|  1.00M|static inline int iclip(const int v, const int min, const int max) {
   52|  1.00M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 57.4k, False: 948k]
  |  Branch (52:28): [True: 11.9k, False: 936k]
  ------------------
   53|  1.00M|}
itx_1d.c:iclip:
   51|   343M|static inline int iclip(const int v, const int min, const int max) {
   52|   343M|    return v < min ? min : v > max ? max : v;
  ------------------
  |  Branch (52:12): [True: 7.04M, False: 336M]
  |  Branch (52:28): [True: 7.01M, False: 329M]
  ------------------
   53|   343M|}
scan.c:imax:
   35|  3.34k|static inline int imax(const int a, const int b) {
   36|  3.34k|    return a > b ? a : b;
  ------------------
  |  Branch (36:12): [True: 2.82k, False: 523]
  ------------------
   37|  3.34k|}
cdef_apply_tmpl.c:imin:
   39|  10.6M|static inline int imin(const int a, const int b) {
   40|  10.6M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 9.41M, False: 1.21M]
  ------------------
   41|  10.6M|}
cdef_apply_tmpl.c:ulog2:
   67|  2.44M|static inline int ulog2(const unsigned v) {
   68|  2.44M|    return 31 ^ clz(v);
   69|  2.44M|}
ipred_prepare_tmpl.c:imin:
   39|  53.6M|static inline int imin(const int a, const int b) {
   40|  53.6M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 50.9M, False: 2.72M]
  ------------------
   41|  53.6M|}
lf_apply_tmpl.c:imin:
   39|  11.9M|static inline int imin(const int a, const int b) {
   40|  11.9M|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 1.25M, False: 10.6M]
  ------------------
   41|  11.9M|}
lr_apply_tmpl.c:imin:
   39|   375k|static inline int imin(const int a, const int b) {
   40|   375k|    return a < b ? a : b;
  ------------------
  |  Branch (40:12): [True: 256k, False: 118k]
  ------------------
   41|   375k|}

dav1d_cdef_brow_8bpc:
  102|   129k|{
  103|   129k|    Dav1dFrameContext *const f = (Dav1dFrameContext *)tc->f;
  104|   129k|    const int bitdepth_min_8 = BITDEPTH == 8 ? 0 : f->cur.p.bpc - 8;
  ------------------
  |  Branch (104:32): [True: 129k, Folded]
  ------------------
  105|   129k|    const Dav1dDSPContext *const dsp = f->dsp;
  106|   129k|    enum CdefEdgeFlags edges = CDEF_HAVE_BOTTOM | (by_start > 0 ? CDEF_HAVE_TOP : 0);
  ------------------
  |  Branch (106:52): [True: 114k, False: 14.7k]
  ------------------
  107|   129k|    pixel *ptrs[3] = { p[0], p[1], p[2] };
  108|   129k|    const int sbsz = 16;
  109|   129k|    const int sb64w = f->sb128w << 1;
  110|   129k|    const int damping = f->frame_hdr->cdef.damping + bitdepth_min_8;
  111|   129k|    const enum Dav1dPixelLayout layout = f->cur.p.layout;
  112|   129k|    const int uv_idx = DAV1D_PIXEL_LAYOUT_I444 - layout;
  113|   129k|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
  114|   129k|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
  115|   129k|    static const uint8_t uv_dirs[2][8] = { { 0, 1, 2, 3, 4, 5, 6, 7 },
  116|   129k|                                           { 7, 0, 2, 4, 5, 6, 6, 6 } };
  117|   129k|    const uint8_t *uv_dir = uv_dirs[layout == DAV1D_PIXEL_LAYOUT_I422];
  118|   129k|    const int have_tt = f->c->n_tc > 1;
  119|   129k|    const int sb128 = f->seq_hdr->sb128;
  120|   129k|    const int resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
  121|   129k|    const ptrdiff_t y_stride = PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|   129k|#define PXSTRIDE(x) (x)
  ------------------
  122|   129k|    const ptrdiff_t uv_stride = PXSTRIDE(f->cur.stride[1]);
  ------------------
  |  |   53|   129k|#define PXSTRIDE(x) (x)
  ------------------
  123|       |
  124|   885k|    for (int bit = 0, by = by_start; by < by_end; by += 2, edges |= CDEF_HAVE_TOP) {
  ------------------
  |  Branch (124:38): [True: 753k, False: 131k]
  ------------------
  125|   753k|        const int tf = tc->top_pre_cdef_toggle;
  126|   753k|        const int by_idx = (by & 30) >> 1;
  127|   753k|        if (by + 2 >= f->bh) edges &= ~CDEF_HAVE_BOTTOM;
  ------------------
  |  Branch (127:13): [True: 14.4k, False: 739k]
  ------------------
  128|       |
  129|   753k|        if ((!have_tt || sbrow_start || by + 2 < by_end) &&
  ------------------
  |  Branch (129:14): [True: 16, False: 753k]
  |  Branch (129:26): [True: 57.2k, False: 696k]
  |  Branch (129:41): [True: 624k, False: 72.3k]
  ------------------
  130|   681k|            edges & CDEF_HAVE_BOTTOM)
  ------------------
  |  Branch (130:13): [True: 682k, False: 18.4E]
  ------------------
  131|   682k|        {
  132|       |            // backup pre-filter data for next iteration
  133|   682k|            pixel *const cdef_top_bak[3] = {
  134|   682k|                f->lf.cdef_line[!tf][0] + have_tt * sby * 4 * y_stride,
  135|   682k|                f->lf.cdef_line[!tf][1] + have_tt * sby * 8 * uv_stride,
  136|   682k|                f->lf.cdef_line[!tf][2] + have_tt * sby * 8 * uv_stride
  137|   682k|            };
  138|   682k|            backup2lines(cdef_top_bak, ptrs, f->cur.stride, layout);
  139|   682k|        }
  140|       |
  141|   753k|        ALIGN_STK_16(pixel, lr_bak, 2 /* idx */, [3 /* plane */][8 /* y */][2 /* x */]);
  ------------------
  |  |  100|   753k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|   753k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  142|   753k|        pixel *iptrs[3] = { ptrs[0], ptrs[1], ptrs[2] };
  143|   753k|        edges &= ~CDEF_HAVE_LEFT;
  144|   753k|        edges |= CDEF_HAVE_RIGHT;
  145|   753k|        enum Backup2x8Flags prev_flag = 0;
  146|  3.06M|        for (int sbx = 0; sbx < sb64w; sbx++, edges |= CDEF_HAVE_LEFT) {
  ------------------
  |  Branch (146:27): [True: 2.30M, False: 755k]
  ------------------
  147|  2.30M|            const int sb128x = sbx >> 1;
  148|  2.30M|            const int sb64_idx = ((by & sbsz) >> 3) + (sbx & 1);
  149|  2.30M|            const int cdef_idx = lflvl[sb128x].cdef_idx[sb64_idx];
  150|  2.30M|            if (cdef_idx == -1 ||
  ------------------
  |  Branch (150:17): [True: 1.43M, False: 874k]
  ------------------
  151|   874k|                (!f->frame_hdr->cdef.y_strength[cdef_idx] &&
  ------------------
  |  Branch (151:18): [True: 375k, False: 499k]
  ------------------
  152|   375k|                 !f->frame_hdr->cdef.uv_strength[cdef_idx]))
  ------------------
  |  Branch (152:18): [True: 316k, False: 58.7k]
  ------------------
  153|  1.75M|            {
  154|  1.75M|                prev_flag = 0;
  155|  1.75M|                goto next_sb;
  156|  1.75M|            }
  157|       |
  158|       |            // Create a complete 32-bit mask for the sb row ahead of time.
  159|   555k|            const uint16_t (*noskip_row)[2] = &lflvl[sb128x].noskip_mask[by_idx];
  160|   555k|            const unsigned noskip_mask = (unsigned) noskip_row[0][1] << 16 |
  161|   555k|                                                    noskip_row[0][0];
  162|       |
  163|   555k|            const int y_lvl = f->frame_hdr->cdef.y_strength[cdef_idx];
  164|   555k|            const int uv_lvl = f->frame_hdr->cdef.uv_strength[cdef_idx];
  165|   555k|            const enum Backup2x8Flags flag = !!y_lvl + (!!uv_lvl << 1);
  166|       |
  167|   555k|            const int y_pri_lvl = (y_lvl >> 2) << bitdepth_min_8;
  168|   555k|            int y_sec_lvl = y_lvl & 3;
  169|   555k|            y_sec_lvl += y_sec_lvl == 3;
  170|   555k|            y_sec_lvl <<= bitdepth_min_8;
  171|       |
  172|   555k|            const int uv_pri_lvl = (uv_lvl >> 2) << bitdepth_min_8;
  173|   555k|            int uv_sec_lvl = uv_lvl & 3;
  174|   555k|            uv_sec_lvl += uv_sec_lvl == 3;
  175|   555k|            uv_sec_lvl <<= bitdepth_min_8;
  176|       |
  177|   555k|            pixel *bptrs[3] = { iptrs[0], iptrs[1], iptrs[2] };
  178|  4.80M|            for (int bx = sbx * sbsz; bx < imin((sbx + 1) * sbsz, f->bw);
  ------------------
  |  Branch (178:39): [True: 4.24M, False: 560k]
  ------------------
  179|  4.24M|                 bx += 2, edges |= CDEF_HAVE_LEFT)
  180|  4.24M|            {
  181|  4.24M|                if (bx + 2 >= f->bw) edges &= ~CDEF_HAVE_RIGHT;
  ------------------
  |  Branch (181:21): [True: 57.8k, False: 4.18M]
  ------------------
  182|       |
  183|       |                // check if this 8x8 block had any coded coefficients; if not,
  184|       |                // go to the next block
  185|  4.24M|                const uint32_t bx_mask = 3U << (bx & 30);
  186|  4.24M|                if (!(noskip_mask & bx_mask)) {
  ------------------
  |  Branch (186:21): [True: 910k, False: 3.33M]
  ------------------
  187|   910k|                    prev_flag = 0;
  188|   910k|                    goto next_b;
  189|   910k|                }
  190|  3.33M|                const enum Backup2x8Flags do_left = (prev_flag ^ flag) & flag;
  191|  3.33M|                prev_flag = flag;
  192|  3.33M|                if (do_left && edges & CDEF_HAVE_LEFT) {
  ------------------
  |  Branch (192:21): [True: 203k, False: 3.13M]
  |  Branch (192:32): [True: 153k, False: 49.9k]
  ------------------
  193|       |                    // we didn't backup the prefilter data because it wasn't
  194|       |                    // there, so do it here instead
  195|   153k|                    backup2x8(lr_bak[bit], bptrs, f->cur.stride, 0, layout, do_left);
  196|   153k|                }
  197|  3.33M|                if (edges & CDEF_HAVE_RIGHT) {
  ------------------
  |  Branch (197:21): [True: 3.29M, False: 40.5k]
  ------------------
  198|       |                    // backup pre-filter data for next iteration
  199|  3.29M|                    backup2x8(lr_bak[!bit], bptrs, f->cur.stride, 8, layout, flag);
  200|  3.29M|                }
  201|       |
  202|  3.33M|                int dir;
  203|  3.33M|                unsigned variance;
  204|  3.33M|                if (y_pri_lvl || uv_pri_lvl)
  ------------------
  |  Branch (204:21): [True: 2.52M, False: 812k]
  |  Branch (204:34): [True: 517k, False: 295k]
  ------------------
  205|  3.01M|                    dir = dsp->cdef.dir(bptrs[0], f->cur.stride[0],
  206|  3.01M|                                        &variance HIGHBD_CALL_SUFFIX);
  207|       |
  208|  3.33M|                const pixel *top, *bot;
  209|  3.33M|                ptrdiff_t offset;
  210|       |
  211|  3.33M|                if (!have_tt) goto st_y;
  ------------------
  |  Branch (211:21): [True: 0, False: 3.33M]
  ------------------
  212|  3.33M|                if (sbrow_start && by == by_start) {
  ------------------
  |  Branch (212:21): [True: 208k, False: 3.12M]
  |  Branch (212:36): [True: 208k, False: 18.4E]
  ------------------
  213|   208k|                    if (resize) {
  ------------------
  |  Branch (213:25): [True: 16.3k, False: 192k]
  ------------------
  214|  16.3k|                        offset = (sby - 1) * 4 * y_stride + bx * 4;
  215|  16.3k|                        top = &f->lf.cdef_lpf_line[0][offset];
  216|   192k|                    } else {
  217|   192k|                        offset = (sby * (4 << sb128) - 4) * y_stride + bx * 4;
  218|   192k|                        top = &f->lf.lr_lpf_line[0][offset];
  219|   192k|                    }
  220|   208k|                    bot = bptrs[0] + 8 * y_stride;
  221|  3.12M|                } else if (!sbrow_start && by + 2 >= by_end) {
  ------------------
  |  Branch (221:28): [True: 3.11M, False: 5.85k]
  |  Branch (221:44): [True: 248k, False: 2.87M]
  ------------------
  222|   248k|                    top = &f->lf.cdef_line[tf][0][sby * 4 * y_stride + bx * 4];
  223|   248k|                    if (resize) {
  ------------------
  |  Branch (223:25): [True: 18.0k, False: 230k]
  ------------------
  224|  18.0k|                        offset = (sby * 4 + 2) * y_stride + bx * 4;
  225|  18.0k|                        bot = &f->lf.cdef_lpf_line[0][offset];
  226|   230k|                    } else {
  227|   230k|                        const int line = sby * (4 << sb128) + 4 * sb128 + 2;
  228|   230k|                        offset = line * y_stride + bx * 4;
  229|   230k|                        bot = &f->lf.lr_lpf_line[0][offset];
  230|   230k|                    }
  231|  2.87M|                } else {
  232|  2.87M|            st_y:;
  233|  2.87M|                    offset = sby * 4 * y_stride;
  234|  2.87M|                    top = &f->lf.cdef_line[tf][0][have_tt * offset + bx * 4];
  235|  2.87M|                    bot = bptrs[0] + 8 * y_stride;
  236|  2.87M|                }
  237|  3.33M|                if (y_pri_lvl) {
  ------------------
  |  Branch (237:21): [True: 2.50M, False: 822k]
  ------------------
  238|  2.50M|                    const int adj_y_pri_lvl = adjust_strength(y_pri_lvl, variance);
  239|  2.50M|                    if (adj_y_pri_lvl || y_sec_lvl)
  ------------------
  |  Branch (239:25): [True: 1.82M, False: 685k]
  |  Branch (239:42): [True: 233k, False: 451k]
  ------------------
  240|  2.05M|                        dsp->cdef.fb[0](bptrs[0], f->cur.stride[0], lr_bak[bit][0],
  241|  2.05M|                                        top, bot, adj_y_pri_lvl, y_sec_lvl,
  242|  2.05M|                                        dir, damping, edges HIGHBD_CALL_SUFFIX);
  243|  2.50M|                } else if (y_sec_lvl)
  ------------------
  |  Branch (243:28): [True: 469k, False: 352k]
  ------------------
  244|   469k|                    dsp->cdef.fb[0](bptrs[0], f->cur.stride[0], lr_bak[bit][0],
  245|   469k|                                    top, bot, 0, y_sec_lvl, 0, damping,
  246|   469k|                                    edges HIGHBD_CALL_SUFFIX);
  247|       |
  248|  3.33M|                if (!uv_lvl) goto skip_uv;
  ------------------
  |  Branch (248:21): [True: 278k, False: 3.05M]
  ------------------
  249|  3.33M|                assert(layout != DAV1D_PIXEL_LAYOUT_I400);
  ------------------
  |  Branch (249:17): [True: 3.04M, False: 3.39k]
  ------------------
  250|       |
  251|  3.04M|                const int uvdir = uv_pri_lvl ? uv_dir[dir] : 0;
  ------------------
  |  Branch (251:35): [True: 2.75M, False: 293k]
  ------------------
  252|  9.11M|                for (int pl = 1; pl <= 2; pl++) {
  ------------------
  |  Branch (252:34): [True: 6.04M, False: 3.06M]
  ------------------
  253|  6.04M|                    if (!have_tt) goto st_uv;
  ------------------
  |  Branch (253:25): [True: 0, False: 6.04M]
  ------------------
  254|  6.04M|                    if (sbrow_start && by == by_start) {
  ------------------
  |  Branch (254:25): [True: 376k, False: 5.66M]
  |  Branch (254:40): [True: 376k, False: 18.4E]
  ------------------
  255|   376k|                        if (resize) {
  ------------------
  |  Branch (255:29): [True: 31.1k, False: 345k]
  ------------------
  256|  31.1k|                            offset = (sby - 1) * 4 * uv_stride + (bx * 4 >> ss_hor);
  257|  31.1k|                            top = &f->lf.cdef_lpf_line[pl][offset];
  258|   345k|                        } else {
  259|   345k|                            const int line = sby * (4 << sb128) - 4;
  260|   345k|                            offset = line * uv_stride + (bx * 4 >> ss_hor);
  261|   345k|                            top = &f->lf.lr_lpf_line[pl][offset];
  262|   345k|                        }
  263|   376k|                        bot = bptrs[pl] + (8 >> ss_ver) * uv_stride;
  264|  5.67M|                    } else if (!sbrow_start && by + 2 >= by_end) {
  ------------------
  |  Branch (264:32): [True: 5.67M, False: 18.4E]
  |  Branch (264:48): [True: 439k, False: 5.23M]
  ------------------
  265|   439k|                        const ptrdiff_t top_offset = sby * 8 * uv_stride +
  266|   439k|                                                     (bx * 4 >> ss_hor);
  267|   439k|                        top = &f->lf.cdef_line[tf][pl][top_offset];
  268|   439k|                        if (resize) {
  ------------------
  |  Branch (268:29): [True: 33.4k, False: 406k]
  ------------------
  269|  33.4k|                            offset = (sby * 4 + 2) * uv_stride + (bx * 4 >> ss_hor);
  270|  33.4k|                            bot = &f->lf.cdef_lpf_line[pl][offset];
  271|   406k|                        } else {
  272|   406k|                            const int line = sby * (4 << sb128) + 4 * sb128 + 2;
  273|   406k|                            offset = line * uv_stride + (bx * 4 >> ss_hor);
  274|   406k|                            bot = &f->lf.lr_lpf_line[pl][offset];
  275|   406k|                        }
  276|  5.22M|                    } else {
  277|  5.24M|                st_uv:;
  278|  5.24M|                        const ptrdiff_t offset = sby * 8 * uv_stride;
  279|  5.24M|                        top = &f->lf.cdef_line[tf][pl][have_tt * offset + (bx * 4 >> ss_hor)];
  280|  5.24M|                        bot = bptrs[pl] + (8 >> ss_ver) * uv_stride;
  281|  5.24M|                    }
  282|  6.06M|                    dsp->cdef.fb[uv_idx](bptrs[pl], f->cur.stride[1],
  283|  6.06M|                                         lr_bak[bit][pl], top, bot,
  284|  6.06M|                                         uv_pri_lvl, uv_sec_lvl, uvdir,
  285|  6.06M|                                         damping - 1, edges HIGHBD_CALL_SUFFIX);
  286|  6.06M|                }
  287|       |
  288|  3.34M|            skip_uv:
  289|  3.34M|                bit ^= 1;
  290|       |
  291|  4.24M|            next_b:
  292|  4.24M|                bptrs[0] += 8;
  293|  4.24M|                bptrs[1] += 8 >> ss_hor;
  294|  4.24M|                bptrs[2] += 8 >> ss_hor;
  295|  4.24M|            }
  296|       |
  297|  2.31M|        next_sb:
  298|  2.31M|            iptrs[0] += sbsz * 4;
  299|  2.31M|            iptrs[1] += sbsz * 4 >> ss_hor;
  300|  2.31M|            iptrs[2] += sbsz * 4 >> ss_hor;
  301|  2.31M|        }
  302|       |
  303|   755k|        ptrs[0] += 8 * PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|   755k|#define PXSTRIDE(x) (x)
  ------------------
  304|   755k|        ptrs[1] += 8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver;
  ------------------
  |  |   53|   755k|#define PXSTRIDE(x) (x)
  ------------------
  305|   755k|        ptrs[2] += 8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver;
  ------------------
  |  |   53|   755k|#define PXSTRIDE(x) (x)
  ------------------
  306|   755k|        tc->top_pre_cdef_toggle ^= 1;
  307|   755k|    }
  308|   129k|}
cdef_apply_tmpl.c:backup2lines:
   44|  12.1M|{
   45|  12.1M|    const ptrdiff_t y_stride = PXSTRIDE(stride[0]);
  ------------------
  |  |   53|  12.1M|#define PXSTRIDE(x) (x)
  ------------------
   46|  12.1M|    if (y_stride < 0)
  ------------------
  |  Branch (46:9): [True: 0, False: 12.1M]
  ------------------
   47|      0|        pixel_copy(dst[0] + y_stride, src[0] + 7 * y_stride, -2 * y_stride);
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
   48|  12.1M|    else
   49|  12.1M|        pixel_copy(dst[0], src[0] + 6 * y_stride, 2 * y_stride);
  ------------------
  |  |   47|  12.1M|#define pixel_copy memcpy
  ------------------
   50|       |
   51|  12.1M|    if (layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (51:9): [True: 1.67M, False: 10.4M]
  ------------------
   52|  1.67M|        const ptrdiff_t uv_stride = PXSTRIDE(stride[1]);
  ------------------
  |  |   53|  1.67M|#define PXSTRIDE(x) (x)
  ------------------
   53|  1.67M|        if (uv_stride < 0) {
  ------------------
  |  Branch (53:13): [True: 0, False: 1.67M]
  ------------------
   54|      0|            const int uv_off = layout == DAV1D_PIXEL_LAYOUT_I420 ? 3 : 7;
  ------------------
  |  Branch (54:32): [True: 0, False: 0]
  ------------------
   55|      0|            pixel_copy(dst[1] + uv_stride, src[1] + uv_off * uv_stride, -2 * uv_stride);
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
   56|      0|            pixel_copy(dst[2] + uv_stride, src[2] + uv_off * uv_stride, -2 * uv_stride);
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
   57|  1.67M|        } else {
   58|  1.67M|            const int uv_off = layout == DAV1D_PIXEL_LAYOUT_I420 ? 2 : 6;
  ------------------
  |  Branch (58:32): [True: 640k, False: 1.03M]
  ------------------
   59|  1.67M|            pixel_copy(dst[1], src[1] + uv_off * uv_stride, 2 * uv_stride);
  ------------------
  |  |   47|  1.67M|#define pixel_copy memcpy
  ------------------
   60|  1.67M|            pixel_copy(dst[2], src[2] + uv_off * uv_stride, 2 * uv_stride);
  ------------------
  |  |   47|  1.67M|#define pixel_copy memcpy
  ------------------
   61|  1.67M|        }
   62|  1.67M|    }
   63|  12.1M|}
cdef_apply_tmpl.c:backup2x8:
   70|  5.95M|{
   71|  5.95M|    ptrdiff_t y_off = 0;
   72|  5.95M|    if (flag & BACKUP_2X8_Y) {
  ------------------
  |  Branch (72:9): [True: 5.52M, False: 431k]
  ------------------
   73|  49.1M|        for (int y = 0; y < 8; y++, y_off += PXSTRIDE(src_stride[0]))
  ------------------
  |  |   53|  43.6M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (73:25): [True: 43.6M, False: 5.52M]
  ------------------
   74|  43.6M|            pixel_copy(dst[0][y], &src[0][y_off + x_off - 2], 2);
  ------------------
  |  |   47|  43.6M|#define pixel_copy memcpy
  ------------------
   75|  5.52M|    }
   76|       |
   77|  5.95M|    if (layout == DAV1D_PIXEL_LAYOUT_I400 || !(flag & BACKUP_2X8_UV))
  ------------------
  |  Branch (77:9): [True: 1.10M, False: 4.85M]
  |  Branch (77:46): [True: 333k, False: 4.51M]
  ------------------
   78|  1.41M|        return;
   79|       |
   80|  4.54M|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
   81|  4.54M|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
   82|       |
   83|  4.54M|    x_off >>= ss_hor;
   84|  4.54M|    y_off = 0;
   85|  25.6M|    for (int y = 0; y < (8 >> ss_ver); y++, y_off += PXSTRIDE(src_stride[1])) {
  ------------------
  |  |   53|  21.1M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (85:21): [True: 21.1M, False: 4.54M]
  ------------------
   86|  21.1M|        pixel_copy(dst[1][y], &src[1][y_off + x_off - 2], 2);
  ------------------
  |  |   47|  21.1M|#define pixel_copy memcpy
  ------------------
   87|  21.1M|        pixel_copy(dst[2][y], &src[2][y_off + x_off - 2], 2);
  ------------------
  |  |   47|  21.1M|#define pixel_copy memcpy
  ------------------
   88|  21.1M|    }
   89|  4.54M|}
cdef_apply_tmpl.c:adjust_strength:
   91|  4.87M|static int adjust_strength(const int strength, const unsigned var) {
   92|  4.87M|    if (!var) return 0;
  ------------------
  |  Branch (92:9): [True: 1.80M, False: 3.06M]
  ------------------
   93|  3.06M|    const int i = var >> 6 ? imin(ulog2(var >> 6), 12) : 0;
  ------------------
  |  Branch (93:19): [True: 2.44M, False: 614k]
  ------------------
   94|  3.06M|    return (strength * (4 + i) + 8) >> 4;
   95|  4.87M|}
dav1d_cdef_brow_16bpc:
  102|  1.57M|{
  103|  1.57M|    Dav1dFrameContext *const f = (Dav1dFrameContext *)tc->f;
  104|  1.57M|    const int bitdepth_min_8 = BITDEPTH == 8 ? 0 : f->cur.p.bpc - 8;
  ------------------
  |  Branch (104:32): [Folded, False: 1.57M]
  ------------------
  105|  1.57M|    const Dav1dDSPContext *const dsp = f->dsp;
  106|  1.57M|    enum CdefEdgeFlags edges = CDEF_HAVE_BOTTOM | (by_start > 0 ? CDEF_HAVE_TOP : 0);
  ------------------
  |  Branch (106:52): [True: 1.42M, False: 143k]
  ------------------
  107|  1.57M|    pixel *ptrs[3] = { p[0], p[1], p[2] };
  108|  1.57M|    const int sbsz = 16;
  109|  1.57M|    const int sb64w = f->sb128w << 1;
  110|  1.57M|    const int damping = f->frame_hdr->cdef.damping + bitdepth_min_8;
  111|  1.57M|    const enum Dav1dPixelLayout layout = f->cur.p.layout;
  112|  1.57M|    const int uv_idx = DAV1D_PIXEL_LAYOUT_I444 - layout;
  113|  1.57M|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
  114|  1.57M|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
  115|  1.57M|    static const uint8_t uv_dirs[2][8] = { { 0, 1, 2, 3, 4, 5, 6, 7 },
  116|  1.57M|                                           { 7, 0, 2, 4, 5, 6, 6, 6 } };
  117|  1.57M|    const uint8_t *uv_dir = uv_dirs[layout == DAV1D_PIXEL_LAYOUT_I422];
  118|  1.57M|    const int have_tt = f->c->n_tc > 1;
  119|  1.57M|    const int sb128 = f->seq_hdr->sb128;
  120|  1.57M|    const int resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
  121|  1.57M|    const ptrdiff_t y_stride = PXSTRIDE(f->cur.stride[0]);
  122|  1.57M|    const ptrdiff_t uv_stride = PXSTRIDE(f->cur.stride[1]);
  123|       |
  124|  13.8M|    for (int bit = 0, by = by_start; by < by_end; by += 2, edges |= CDEF_HAVE_TOP) {
  ------------------
  |  Branch (124:38): [True: 12.2M, False: 1.58M]
  ------------------
  125|  12.2M|        const int tf = tc->top_pre_cdef_toggle;
  126|  12.2M|        const int by_idx = (by & 30) >> 1;
  127|  12.2M|        if (by + 2 >= f->bh) edges &= ~CDEF_HAVE_BOTTOM;
  ------------------
  |  Branch (127:13): [True: 143k, False: 12.1M]
  ------------------
  128|       |
  129|  12.2M|        if ((!have_tt || sbrow_start || by + 2 < by_end) &&
  ------------------
  |  Branch (129:14): [True: 18.4E, False: 12.2M]
  |  Branch (129:26): [True: 711k, False: 11.5M]
  |  Branch (129:41): [True: 10.7M, False: 859k]
  ------------------
  130|  11.4M|            edges & CDEF_HAVE_BOTTOM)
  ------------------
  |  Branch (130:13): [True: 11.4M, False: 18.4E]
  ------------------
  131|  11.4M|        {
  132|       |            // backup pre-filter data for next iteration
  133|  11.4M|            pixel *const cdef_top_bak[3] = {
  134|  11.4M|                f->lf.cdef_line[!tf][0] + have_tt * sby * 4 * y_stride,
  135|  11.4M|                f->lf.cdef_line[!tf][1] + have_tt * sby * 8 * uv_stride,
  136|  11.4M|                f->lf.cdef_line[!tf][2] + have_tt * sby * 8 * uv_stride
  137|  11.4M|            };
  138|  11.4M|            backup2lines(cdef_top_bak, ptrs, f->cur.stride, layout);
  139|  11.4M|        }
  140|       |
  141|  12.2M|        ALIGN_STK_16(pixel, lr_bak, 2 /* idx */, [3 /* plane */][8 /* y */][2 /* x */]);
  ------------------
  |  |  100|  12.2M|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  12.2M|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  142|  12.2M|        pixel *iptrs[3] = { ptrs[0], ptrs[1], ptrs[2] };
  143|  12.2M|        edges &= ~CDEF_HAVE_LEFT;
  144|  12.2M|        edges |= CDEF_HAVE_RIGHT;
  145|  12.2M|        enum Backup2x8Flags prev_flag = 0;
  146|  37.3M|        for (int sbx = 0; sbx < sb64w; sbx++, edges |= CDEF_HAVE_LEFT) {
  ------------------
  |  Branch (146:27): [True: 25.0M, False: 12.2M]
  ------------------
  147|  25.0M|            const int sb128x = sbx >> 1;
  148|  25.0M|            const int sb64_idx = ((by & sbsz) >> 3) + (sbx & 1);
  149|  25.0M|            const int cdef_idx = lflvl[sb128x].cdef_idx[sb64_idx];
  150|  25.0M|            if (cdef_idx == -1 ||
  ------------------
  |  Branch (150:17): [True: 23.4M, False: 1.58M]
  ------------------
  151|  1.58M|                (!f->frame_hdr->cdef.y_strength[cdef_idx] &&
  ------------------
  |  Branch (151:18): [True: 1.11M, False: 471k]
  ------------------
  152|  1.11M|                 !f->frame_hdr->cdef.uv_strength[cdef_idx]))
  ------------------
  |  Branch (152:18): [True: 1.09M, False: 16.3k]
  ------------------
  153|  24.5M|            {
  154|  24.5M|                prev_flag = 0;
  155|  24.5M|                goto next_sb;
  156|  24.5M|            }
  157|       |
  158|       |            // Create a complete 32-bit mask for the sb row ahead of time.
  159|   463k|            const uint16_t (*noskip_row)[2] = &lflvl[sb128x].noskip_mask[by_idx];
  160|   463k|            const unsigned noskip_mask = (unsigned) noskip_row[0][1] << 16 |
  161|   463k|                                                    noskip_row[0][0];
  162|       |
  163|   463k|            const int y_lvl = f->frame_hdr->cdef.y_strength[cdef_idx];
  164|   463k|            const int uv_lvl = f->frame_hdr->cdef.uv_strength[cdef_idx];
  165|   463k|            const enum Backup2x8Flags flag = !!y_lvl + (!!uv_lvl << 1);
  166|       |
  167|   463k|            const int y_pri_lvl = (y_lvl >> 2) << bitdepth_min_8;
  168|   463k|            int y_sec_lvl = y_lvl & 3;
  169|   463k|            y_sec_lvl += y_sec_lvl == 3;
  170|   463k|            y_sec_lvl <<= bitdepth_min_8;
  171|       |
  172|   463k|            const int uv_pri_lvl = (uv_lvl >> 2) << bitdepth_min_8;
  173|   463k|            int uv_sec_lvl = uv_lvl & 3;
  174|   463k|            uv_sec_lvl += uv_sec_lvl == 3;
  175|   463k|            uv_sec_lvl <<= bitdepth_min_8;
  176|       |
  177|   463k|            pixel *bptrs[3] = { iptrs[0], iptrs[1], iptrs[2] };
  178|  3.37M|            for (int bx = sbx * sbsz; bx < imin((sbx + 1) * sbsz, f->bw);
  ------------------
  |  Branch (178:39): [True: 2.91M, False: 463k]
  ------------------
  179|  2.91M|                 bx += 2, edges |= CDEF_HAVE_LEFT)
  180|  2.91M|            {
  181|  2.91M|                if (bx + 2 >= f->bw) edges &= ~CDEF_HAVE_RIGHT;
  ------------------
  |  Branch (181:21): [True: 172k, False: 2.74M]
  ------------------
  182|       |
  183|       |                // check if this 8x8 block had any coded coefficients; if not,
  184|       |                // go to the next block
  185|  2.91M|                const uint32_t bx_mask = 3U << (bx & 30);
  186|  2.91M|                if (!(noskip_mask & bx_mask)) {
  ------------------
  |  Branch (186:21): [True: 278k, False: 2.63M]
  ------------------
  187|   278k|                    prev_flag = 0;
  188|   278k|                    goto next_b;
  189|   278k|                }
  190|  2.63M|                const enum Backup2x8Flags do_left = (prev_flag ^ flag) & flag;
  191|  2.63M|                prev_flag = flag;
  192|  2.63M|                if (do_left && edges & CDEF_HAVE_LEFT) {
  ------------------
  |  Branch (192:21): [True: 211k, False: 2.42M]
  |  Branch (192:32): [True: 40.6k, False: 170k]
  ------------------
  193|       |                    // we didn't backup the prefilter data because it wasn't
  194|       |                    // there, so do it here instead
  195|  40.6k|                    backup2x8(lr_bak[bit], bptrs, f->cur.stride, 0, layout, do_left);
  196|  40.6k|                }
  197|  2.63M|                if (edges & CDEF_HAVE_RIGHT) {
  ------------------
  |  Branch (197:21): [True: 2.46M, False: 166k]
  ------------------
  198|       |                    // backup pre-filter data for next iteration
  199|  2.46M|                    backup2x8(lr_bak[!bit], bptrs, f->cur.stride, 8, layout, flag);
  200|  2.46M|                }
  201|       |
  202|  2.63M|                int dir;
  203|  2.63M|                unsigned variance;
  204|  2.63M|                if (y_pri_lvl || uv_pri_lvl)
  ------------------
  |  Branch (204:21): [True: 2.36M, False: 269k]
  |  Branch (204:34): [True: 90.4k, False: 179k]
  ------------------
  205|  2.44M|                    dir = dsp->cdef.dir(bptrs[0], f->cur.stride[0],
  206|  2.44M|                                        &variance HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  2.44M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
  207|       |
  208|  2.63M|                const pixel *top, *bot;
  209|  2.63M|                ptrdiff_t offset;
  210|       |
  211|  2.63M|                if (!have_tt) goto st_y;
  ------------------
  |  Branch (211:21): [True: 0, False: 2.63M]
  ------------------
  212|  2.63M|                if (sbrow_start && by == by_start) {
  ------------------
  |  Branch (212:21): [True: 156k, False: 2.47M]
  |  Branch (212:36): [True: 156k, False: 18.4E]
  ------------------
  213|   156k|                    if (resize) {
  ------------------
  |  Branch (213:25): [True: 48.2k, False: 108k]
  ------------------
  214|  48.2k|                        offset = (sby - 1) * 4 * y_stride + bx * 4;
  215|  48.2k|                        top = &f->lf.cdef_lpf_line[0][offset];
  216|   108k|                    } else {
  217|   108k|                        offset = (sby * (4 << sb128) - 4) * y_stride + bx * 4;
  218|   108k|                        top = &f->lf.lr_lpf_line[0][offset];
  219|   108k|                    }
  220|   156k|                    bot = bptrs[0] + 8 * y_stride;
  221|  2.48M|                } else if (!sbrow_start && by + 2 >= by_end) {
  ------------------
  |  Branch (221:28): [True: 2.48M, False: 18.4E]
  |  Branch (221:44): [True: 222k, False: 2.25M]
  ------------------
  222|   222k|                    top = &f->lf.cdef_line[tf][0][sby * 4 * y_stride + bx * 4];
  223|   222k|                    if (resize) {
  ------------------
  |  Branch (223:25): [True: 66.8k, False: 155k]
  ------------------
  224|  66.8k|                        offset = (sby * 4 + 2) * y_stride + bx * 4;
  225|  66.8k|                        bot = &f->lf.cdef_lpf_line[0][offset];
  226|   155k|                    } else {
  227|   155k|                        const int line = sby * (4 << sb128) + 4 * sb128 + 2;
  228|   155k|                        offset = line * y_stride + bx * 4;
  229|   155k|                        bot = &f->lf.lr_lpf_line[0][offset];
  230|   155k|                    }
  231|  2.25M|                } else {
  232|  2.26M|            st_y:;
  233|  2.26M|                    offset = sby * 4 * y_stride;
  234|  2.26M|                    top = &f->lf.cdef_line[tf][0][have_tt * offset + bx * 4];
  235|  2.26M|                    bot = bptrs[0] + 8 * y_stride;
  236|  2.26M|                }
  237|  2.64M|                if (y_pri_lvl) {
  ------------------
  |  Branch (237:21): [True: 2.36M, False: 280k]
  ------------------
  238|  2.36M|                    const int adj_y_pri_lvl = adjust_strength(y_pri_lvl, variance);
  239|  2.36M|                    if (adj_y_pri_lvl || y_sec_lvl)
  ------------------
  |  Branch (239:25): [True: 1.01M, False: 1.34M]
  |  Branch (239:42): [True: 518k, False: 829k]
  ------------------
  240|  1.52M|                        dsp->cdef.fb[0](bptrs[0], f->cur.stride[0], lr_bak[bit][0],
  241|  1.52M|                                        top, bot, adj_y_pri_lvl, y_sec_lvl,
  242|  1.52M|                                        dir, damping, edges HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  1.52M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
  243|  2.36M|                } else if (y_sec_lvl)
  ------------------
  |  Branch (243:28): [True: 198k, False: 82.3k]
  ------------------
  244|   198k|                    dsp->cdef.fb[0](bptrs[0], f->cur.stride[0], lr_bak[bit][0],
  245|   198k|                                    top, bot, 0, y_sec_lvl, 0, damping,
  246|   198k|                                    edges HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   198k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
  247|       |
  248|  2.64M|                if (!uv_lvl) goto skip_uv;
  ------------------
  |  Branch (248:21): [True: 1.20M, False: 1.44M]
  ------------------
  249|  2.64M|                assert(layout != DAV1D_PIXEL_LAYOUT_I400);
  ------------------
  |  Branch (249:17): [True: 1.43M, False: 7.87k]
  ------------------
  250|       |
  251|  1.43M|                const int uvdir = uv_pri_lvl ? uv_dir[dir] : 0;
  ------------------
  |  Branch (251:35): [True: 1.37M, False: 53.1k]
  ------------------
  252|  4.28M|                for (int pl = 1; pl <= 2; pl++) {
  ------------------
  |  Branch (252:34): [True: 2.84M, False: 1.44M]
  ------------------
  253|  2.84M|                    if (!have_tt) goto st_uv;
  ------------------
  |  Branch (253:25): [True: 0, False: 2.84M]
  ------------------
  254|  2.84M|                    if (sbrow_start && by == by_start) {
  ------------------
  |  Branch (254:25): [True: 168k, False: 2.67M]
  |  Branch (254:40): [True: 168k, False: 18.4E]
  ------------------
  255|   168k|                        if (resize) {
  ------------------
  |  Branch (255:29): [True: 48.8k, False: 119k]
  ------------------
  256|  48.8k|                            offset = (sby - 1) * 4 * uv_stride + (bx * 4 >> ss_hor);
  257|  48.8k|                            top = &f->lf.cdef_lpf_line[pl][offset];
  258|   119k|                        } else {
  259|   119k|                            const int line = sby * (4 << sb128) - 4;
  260|   119k|                            offset = line * uv_stride + (bx * 4 >> ss_hor);
  261|   119k|                            top = &f->lf.lr_lpf_line[pl][offset];
  262|   119k|                        }
  263|   168k|                        bot = bptrs[pl] + (8 >> ss_ver) * uv_stride;
  264|  2.68M|                    } else if (!sbrow_start && by + 2 >= by_end) {
  ------------------
  |  Branch (264:32): [True: 2.68M, False: 18.4E]
  |  Branch (264:48): [True: 217k, False: 2.46M]
  ------------------
  265|   217k|                        const ptrdiff_t top_offset = sby * 8 * uv_stride +
  266|   217k|                                                     (bx * 4 >> ss_hor);
  267|   217k|                        top = &f->lf.cdef_line[tf][pl][top_offset];
  268|   217k|                        if (resize) {
  ------------------
  |  Branch (268:29): [True: 57.4k, False: 160k]
  ------------------
  269|  57.4k|                            offset = (sby * 4 + 2) * uv_stride + (bx * 4 >> ss_hor);
  270|  57.4k|                            bot = &f->lf.cdef_lpf_line[pl][offset];
  271|   160k|                        } else {
  272|   160k|                            const int line = sby * (4 << sb128) + 4 * sb128 + 2;
  273|   160k|                            offset = line * uv_stride + (bx * 4 >> ss_hor);
  274|   160k|                            bot = &f->lf.lr_lpf_line[pl][offset];
  275|   160k|                        }
  276|  2.45M|                    } else {
  277|  2.46M|                st_uv:;
  278|  2.46M|                        const ptrdiff_t offset = sby * 8 * uv_stride;
  279|  2.46M|                        top = &f->lf.cdef_line[tf][pl][have_tt * offset + (bx * 4 >> ss_hor)];
  280|  2.46M|                        bot = bptrs[pl] + (8 >> ss_ver) * uv_stride;
  281|  2.46M|                    }
  282|  2.85M|                    dsp->cdef.fb[uv_idx](bptrs[pl], f->cur.stride[1],
  283|  2.85M|                                         lr_bak[bit][pl], top, bot,
  284|  2.85M|                                         uv_pri_lvl, uv_sec_lvl, uvdir,
  285|  2.85M|                                         damping - 1, edges HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  2.85M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
  286|  2.85M|                }
  287|       |
  288|  2.63M|            skip_uv:
  289|  2.63M|                bit ^= 1;
  290|       |
  291|  2.91M|            next_b:
  292|  2.91M|                bptrs[0] += 8;
  293|  2.91M|                bptrs[1] += 8 >> ss_hor;
  294|  2.91M|                bptrs[2] += 8 >> ss_hor;
  295|  2.91M|            }
  296|       |
  297|  25.0M|        next_sb:
  298|  25.0M|            iptrs[0] += sbsz * 4;
  299|  25.0M|            iptrs[1] += sbsz * 4 >> ss_hor;
  300|  25.0M|            iptrs[2] += sbsz * 4 >> ss_hor;
  301|  25.0M|        }
  302|       |
  303|  12.2M|        ptrs[0] += 8 * PXSTRIDE(f->cur.stride[0]);
  304|  12.2M|        ptrs[1] += 8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver;
  305|  12.2M|        ptrs[2] += 8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver;
  306|  12.2M|        tc->top_pre_cdef_toggle ^= 1;
  307|  12.2M|    }
  308|  1.57M|}

dav1d_cdef_dsp_init_8bpc:
  321|  3.46k|COLD void bitfn(dav1d_cdef_dsp_init)(Dav1dCdefDSPContext *const c) {
  322|  3.46k|    c->dir = cdef_find_dir_c;
  323|  3.46k|    c->fb[0] = cdef_filter_block_8x8_c;
  324|  3.46k|    c->fb[1] = cdef_filter_block_4x8_c;
  325|  3.46k|    c->fb[2] = cdef_filter_block_4x4_c;
  326|       |
  327|  3.46k|#if HAVE_ASM
  328|       |#if ARCH_AARCH64 || ARCH_ARM
  329|       |    cdef_dsp_init_arm(c);
  330|       |#elif ARCH_PPC64LE
  331|       |    cdef_dsp_init_ppc(c);
  332|       |#elif ARCH_RISCV
  333|       |    cdef_dsp_init_riscv(c);
  334|       |#elif ARCH_X86
  335|       |    cdef_dsp_init_x86(c);
  336|       |#elif ARCH_LOONGARCH64
  337|       |    cdef_dsp_init_loongarch(c);
  338|       |#endif
  339|  3.46k|#endif
  340|  3.46k|}
dav1d_cdef_dsp_init_16bpc:
  321|  5.11k|COLD void bitfn(dav1d_cdef_dsp_init)(Dav1dCdefDSPContext *const c) {
  322|  5.11k|    c->dir = cdef_find_dir_c;
  323|  5.11k|    c->fb[0] = cdef_filter_block_8x8_c;
  324|  5.11k|    c->fb[1] = cdef_filter_block_4x8_c;
  325|  5.11k|    c->fb[2] = cdef_filter_block_4x4_c;
  326|       |
  327|  5.11k|#if HAVE_ASM
  328|       |#if ARCH_AARCH64 || ARCH_ARM
  329|       |    cdef_dsp_init_arm(c);
  330|       |#elif ARCH_PPC64LE
  331|       |    cdef_dsp_init_ppc(c);
  332|       |#elif ARCH_RISCV
  333|       |    cdef_dsp_init_riscv(c);
  334|       |#elif ARCH_X86
  335|       |    cdef_dsp_init_x86(c);
  336|       |#elif ARCH_LOONGARCH64
  337|       |    cdef_dsp_init_loongarch(c);
  338|       |#endif
  339|  5.11k|#endif
  340|  5.11k|}

dav1d_cdf_thread_update:
 3918|  9.85k|{
 3919|  9.85k|#define update_cdf_1d(n1d, name) \
 3920|  9.85k|    do { \
 3921|  9.85k|        dst->name[n1d] = 0; \
 3922|  9.85k|    } while (0)
 3923|  9.85k|#define update_cdf_2d(n1d, n2d, name) \
 3924|  9.85k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
 3925|  9.85k|#define update_cdf_3d(n1d, n2d, n3d, name) \
 3926|  9.85k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
 3927|  9.85k|#define update_cdf_4d(n1d, n2d, n3d, n4d, name) \
 3928|  9.85k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
 3929|       |
 3930|  9.85k|    memcpy(dst, src, offsetof(CdfContext, m.intrabc));
 3931|       |
 3932|  9.85k|    update_cdf_3d(2, 2, 4, coef.eob_bin_16);
  ------------------
  |  | 3926|  29.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  59.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  39.4k|    do { \
  |  |  |  |  |  | 3921|  39.4k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  39.4k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 39.4k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 39.4k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3933|  9.85k|    update_cdf_3d(2, 2, 5, coef.eob_bin_32);
  ------------------
  |  | 3926|  29.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  59.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  39.4k|    do { \
  |  |  |  |  |  | 3921|  39.4k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  39.4k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 39.4k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 39.4k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3934|  9.85k|    update_cdf_3d(2, 2, 6, coef.eob_bin_64);
  ------------------
  |  | 3926|  29.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  59.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  39.4k|    do { \
  |  |  |  |  |  | 3921|  39.4k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  39.4k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 39.4k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 39.4k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3935|  9.85k|    update_cdf_3d(2, 2, 7, coef.eob_bin_128);
  ------------------
  |  | 3926|  29.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  59.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  39.4k|    do { \
  |  |  |  |  |  | 3921|  39.4k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  39.4k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 39.4k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 39.4k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3936|  9.85k|    update_cdf_3d(2, 2, 8, coef.eob_bin_256);
  ------------------
  |  | 3926|  29.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  59.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  39.4k|    do { \
  |  |  |  |  |  | 3921|  39.4k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  39.4k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 39.4k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 39.4k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3937|  9.85k|    update_cdf_2d(2, 9, coef.eob_bin_512);
  ------------------
  |  | 3924|  29.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  19.7k|    do { \
  |  |  |  | 3921|  19.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  19.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3938|  9.85k|    update_cdf_2d(2, 10, coef.eob_bin_1024);
  ------------------
  |  | 3924|  29.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  19.7k|    do { \
  |  |  |  | 3921|  19.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  19.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3939|  9.85k|    update_cdf_4d(N_TX_SIZES, 2, 4, 2, coef.eob_base_tok);
  ------------------
  |  | 3928|  59.1k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   147k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|   492k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|   394k|    do { \
  |  |  |  |  |  |  |  | 3921|   394k|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|   394k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 394k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 394k, False: 98.5k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 98.5k, False: 49.2k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 49.2k, False: 9.85k]
  |  |  ------------------
  ------------------
 3940|  9.85k|    update_cdf_4d(N_TX_SIZES, 2, 41 /*42*/, 3, coef.base_tok);
  ------------------
  |  | 3928|  59.1k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   147k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|  4.13M|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|  4.03M|    do { \
  |  |  |  |  |  |  |  | 3921|  4.03M|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|  4.03M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 4.03M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 4.03M, False: 98.5k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 98.5k, False: 49.2k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 49.2k, False: 9.85k]
  |  |  ------------------
  ------------------
 3941|  9.85k|    update_cdf_4d(4, 2, 21, 3, coef.br_tok);
  ------------------
  |  | 3928|  49.2k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   118k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|  1.73M|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|  1.65M|    do { \
  |  |  |  |  |  |  |  | 3921|  1.65M|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|  1.65M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 1.65M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 1.65M, False: 78.8k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 78.8k, False: 39.4k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 39.4k, False: 9.85k]
  |  |  ------------------
  ------------------
 3942|  9.85k|    update_cdf_4d(N_TX_SIZES, 2, 9, 1, coef.eob_hi_bit);
  ------------------
  |  | 3928|  59.1k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   147k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|   985k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|   886k|    do { \
  |  |  |  |  |  |  |  | 3921|   886k|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|   886k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 886k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 886k, False: 98.5k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 98.5k, False: 49.2k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 49.2k, False: 9.85k]
  |  |  ------------------
  ------------------
 3943|  9.85k|    update_cdf_3d(N_TX_SIZES, 13, 1, coef.skip);
  ------------------
  |  | 3926|  59.1k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   689k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   640k|    do { \
  |  |  |  |  |  | 3921|   640k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   640k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 640k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 640k, False: 49.2k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 49.2k, False: 9.85k]
  |  |  ------------------
  ------------------
 3944|  9.85k|    update_cdf_3d(2, 3, 1, coef.dc_sign);
  ------------------
  |  | 3926|  29.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  78.8k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  59.1k|    do { \
  |  |  |  |  |  | 3921|  59.1k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  59.1k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 59.1k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 59.1k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3945|       |
 3946|  9.85k|    update_cdf_3d(2, N_INTRA_PRED_MODES, N_UV_INTRA_PRED_MODES - 1 - !k, m.uv_mode);
  ------------------
  |  | 3926|  29.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   275k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   256k|    do { \
  |  |  |  |  |  | 3921|   256k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   256k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 256k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 256k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3947|  9.85k|    update_cdf_2d(4, N_PARTITIONS - 3, m.partition[BL_128X128]);
  ------------------
  |  | 3924|  49.2k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  39.4k|    do { \
  |  |  |  | 3921|  39.4k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  39.4k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 39.4k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 39.4k, False: 9.85k]
  |  |  ------------------
  ------------------
 3948|  39.4k|    for (int k = BL_64X64; k < BL_8X8; k++)
  ------------------
  |  Branch (3948:28): [True: 29.5k, False: 9.85k]
  ------------------
 3949|  29.5k|        update_cdf_2d(4, N_PARTITIONS - 1, m.partition[k]);
  ------------------
  |  | 3924|   147k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|   118k|    do { \
  |  |  |  | 3921|   118k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|   118k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 118k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 118k, False: 29.5k]
  |  |  ------------------
  ------------------
 3950|  9.85k|    update_cdf_2d(4, N_SUB8X8_PARTITIONS - 1, m.partition[BL_8X8]);
  ------------------
  |  | 3924|  49.2k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  39.4k|    do { \
  |  |  |  | 3921|  39.4k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  39.4k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 39.4k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 39.4k, False: 9.85k]
  |  |  ------------------
  ------------------
 3951|  9.85k|    update_cdf_2d(6, 15, m.cfl_alpha);
  ------------------
  |  | 3924|  68.9k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  59.1k|    do { \
  |  |  |  | 3921|  59.1k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  59.1k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 59.1k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 59.1k, False: 9.85k]
  |  |  ------------------
  ------------------
 3952|  9.85k|    update_cdf_2d(2, 15, m.txtp_inter1);
  ------------------
  |  | 3924|  29.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  19.7k|    do { \
  |  |  |  | 3921|  19.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  19.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3953|  9.85k|    update_cdf_1d(11, m.txtp_inter2);
  ------------------
  |  | 3920|  9.85k|    do { \
  |  | 3921|  9.85k|        dst->name[n1d] = 0; \
  |  | 3922|  9.85k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 9.85k]
  |  |  ------------------
  ------------------
 3954|  9.85k|    update_cdf_3d(2, N_INTRA_PRED_MODES, 6, m.txtp_intra1);
  ------------------
  |  | 3926|  29.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   275k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   256k|    do { \
  |  |  |  |  |  | 3921|   256k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   256k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 256k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 256k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3955|  9.85k|    update_cdf_3d(3, N_INTRA_PRED_MODES, 4, m.txtp_intra2);
  ------------------
  |  | 3926|  39.4k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   413k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   384k|    do { \
  |  |  |  |  |  | 3921|   384k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   384k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 384k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 384k, False: 29.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 29.5k, False: 9.85k]
  |  |  ------------------
  ------------------
 3956|  9.85k|    update_cdf_1d(7, m.cfl_sign);
  ------------------
  |  | 3920|  9.85k|    do { \
  |  | 3921|  9.85k|        dst->name[n1d] = 0; \
  |  | 3922|  9.85k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 9.85k]
  |  |  ------------------
  ------------------
 3957|  9.85k|    update_cdf_2d(8, 6, m.angle_delta);
  ------------------
  |  | 3924|  88.6k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  78.8k|    do { \
  |  |  |  | 3921|  78.8k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  78.8k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 78.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 78.8k, False: 9.85k]
  |  |  ------------------
  ------------------
 3958|  9.85k|    update_cdf_1d(4, m.filter_intra);
  ------------------
  |  | 3920|  9.85k|    do { \
  |  | 3921|  9.85k|        dst->name[n1d] = 0; \
  |  | 3922|  9.85k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 9.85k]
  |  |  ------------------
  ------------------
 3959|  9.85k|    update_cdf_2d(3, DAV1D_MAX_SEGMENTS - 1, m.seg_id);
  ------------------
  |  | 3924|  39.4k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  29.5k|    do { \
  |  |  |  | 3921|  29.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  29.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 29.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 29.5k, False: 9.85k]
  |  |  ------------------
  ------------------
 3960|  9.85k|    update_cdf_3d(2, 7, 6, m.pal_sz);
  ------------------
  |  | 3926|  29.5k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   157k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   137k|    do { \
  |  |  |  |  |  | 3921|   137k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   137k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 137k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 137k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3961|  9.85k|    update_cdf_4d(2, 7, 5, k + 1, m.color_map);
  ------------------
  |  | 3928|  29.5k|    for (int l = 0; l < (n1d); l++) update_cdf_3d(n2d, n3d, n4d, name[l])
  |  |  ------------------
  |  |  |  | 3926|   157k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3924|   827k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  | 3920|   689k|    do { \
  |  |  |  |  |  |  |  | 3921|   689k|        dst->name[n1d] = 0; \
  |  |  |  |  |  |  |  | 3922|   689k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 689k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3924:21): [True: 689k, False: 137k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3926:21): [True: 137k, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3928:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3962|  9.85k|    update_cdf_3d(N_TX_SIZES - 1, 3, imin(k + 1, 2), m.txsz);
  ------------------
  |  | 3926|  49.2k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   157k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   118k|    do { \
  |  |  |  |  |  | 3921|   118k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   118k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 118k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 118k, False: 39.4k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 39.4k, False: 9.85k]
  |  |  ------------------
  ------------------
 3963|  9.85k|    update_cdf_1d(3, m.delta_q);
  ------------------
  |  | 3920|  9.85k|    do { \
  |  | 3921|  9.85k|        dst->name[n1d] = 0; \
  |  | 3922|  9.85k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 9.85k]
  |  |  ------------------
  ------------------
 3964|  9.85k|    update_cdf_2d(5, 3, m.delta_lf);
  ------------------
  |  | 3924|  59.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  49.2k|    do { \
  |  |  |  | 3921|  49.2k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  49.2k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 49.2k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 49.2k, False: 9.85k]
  |  |  ------------------
  ------------------
 3965|  9.85k|    update_cdf_1d(2, m.restore_switchable);
  ------------------
  |  | 3920|  9.85k|    do { \
  |  | 3921|  9.85k|        dst->name[n1d] = 0; \
  |  | 3922|  9.85k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 9.85k]
  |  |  ------------------
  ------------------
 3966|  9.85k|    update_cdf_1d(1, m.restore_wiener);
  ------------------
  |  | 3920|  9.85k|    do { \
  |  | 3921|  9.85k|        dst->name[n1d] = 0; \
  |  | 3922|  9.85k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 9.85k]
  |  |  ------------------
  ------------------
 3967|  9.85k|    update_cdf_1d(1, m.restore_sgrproj);
  ------------------
  |  | 3920|  9.85k|    do { \
  |  | 3921|  9.85k|        dst->name[n1d] = 0; \
  |  | 3922|  9.85k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 9.85k]
  |  |  ------------------
  ------------------
 3968|  9.85k|    update_cdf_2d(4, 1, m.txtp_inter3);
  ------------------
  |  | 3924|  49.2k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  39.4k|    do { \
  |  |  |  | 3921|  39.4k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  39.4k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 39.4k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 39.4k, False: 9.85k]
  |  |  ------------------
  ------------------
 3969|  9.85k|    update_cdf_2d(N_BS_SIZES, 1, m.use_filter_intra);
  ------------------
  |  | 3924|   226k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|   216k|    do { \
  |  |  |  | 3921|   216k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|   216k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 216k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 216k, False: 9.85k]
  |  |  ------------------
  ------------------
 3970|  9.85k|    update_cdf_3d(7, 3, 1, m.txpart);
  ------------------
  |  | 3926|  78.8k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   275k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   206k|    do { \
  |  |  |  |  |  | 3921|   206k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   206k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 206k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 206k, False: 68.9k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 68.9k, False: 9.85k]
  |  |  ------------------
  ------------------
 3971|  9.85k|    update_cdf_2d(3, 1, m.skip);
  ------------------
  |  | 3924|  39.4k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  29.5k|    do { \
  |  |  |  | 3921|  29.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  29.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 29.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 29.5k, False: 9.85k]
  |  |  ------------------
  ------------------
 3972|  9.85k|    update_cdf_3d(7, 3, 1, m.pal_y);
  ------------------
  |  | 3926|  78.8k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   275k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|   206k|    do { \
  |  |  |  |  |  | 3921|   206k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|   206k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 206k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 206k, False: 68.9k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 68.9k, False: 9.85k]
  |  |  ------------------
  ------------------
 3973|  9.85k|    update_cdf_2d(2, 1, m.pal_uv);
  ------------------
  |  | 3924|  29.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  19.7k|    do { \
  |  |  |  | 3921|  19.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  19.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 19.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 19.7k, False: 9.85k]
  |  |  ------------------
  ------------------
 3974|       |
 3975|  9.85k|    if (IS_KEY_OR_INTRA(hdr))
  ------------------
  |  |   43|  9.85k|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  9.85k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 5.59k, False: 4.26k]
  |  |  ------------------
  ------------------
 3976|  5.59k|        return;
 3977|       |
 3978|  4.26k|    memcpy(dst->m.y_mode, src->m.y_mode,
 3979|  4.26k|           offsetof(CdfContext, kfym) - offsetof(CdfContext, m.y_mode));
 3980|       |
 3981|  4.26k|    update_cdf_2d(4, N_INTRA_PRED_MODES - 1, m.y_mode);
  ------------------
  |  | 3924|  21.3k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  21.3k|    do { \
  |  |  |  | 3921|  17.0k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  17.0k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 17.0k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 17.0k, False: 4.26k]
  |  |  ------------------
  ------------------
 3982|  4.26k|    update_cdf_2d(9, 15, m.wedge_idx);
  ------------------
  |  | 3924|  42.6k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  42.6k|    do { \
  |  |  |  | 3921|  38.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  38.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 38.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 38.3k, False: 4.26k]
  |  |  ------------------
  ------------------
 3983|  4.26k|    update_cdf_2d(8, N_COMP_INTER_PRED_MODES - 1, m.comp_inter_mode);
  ------------------
  |  | 3924|  38.3k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  38.3k|    do { \
  |  |  |  | 3921|  34.0k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  34.0k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 34.0k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 34.0k, False: 4.26k]
  |  |  ------------------
  ------------------
 3984|  4.26k|    update_cdf_3d(2, 8, DAV1D_N_SWITCHABLE_FILTERS - 1, m.filter);
  ------------------
  |  | 3926|  12.7k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  76.7k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  72.4k|    do { \
  |  |  |  |  |  | 3921|  68.1k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  68.1k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 68.1k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 68.1k, False: 8.52k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 8.52k, False: 4.26k]
  |  |  ------------------
  ------------------
 3985|  4.26k|    update_cdf_2d(4, 3, m.interintra_mode);
  ------------------
  |  | 3924|  21.3k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  21.3k|    do { \
  |  |  |  | 3921|  17.0k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  17.0k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 17.0k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 17.0k, False: 4.26k]
  |  |  ------------------
  ------------------
 3986|  4.26k|    update_cdf_2d(N_BS_SIZES, 2, m.motion_mode);
  ------------------
  |  | 3924|  98.0k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  98.0k|    do { \
  |  |  |  | 3921|  93.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  93.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 93.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 93.7k, False: 4.26k]
  |  |  ------------------
  ------------------
 3987|  4.26k|    update_cdf_2d(3, 1, m.skip_mode);
  ------------------
  |  | 3924|  17.0k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  17.0k|    do { \
  |  |  |  | 3921|  12.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  12.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 12.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 12.7k, False: 4.26k]
  |  |  ------------------
  ------------------
 3988|  4.26k|    update_cdf_2d(6, 1, m.newmv_mode);
  ------------------
  |  | 3924|  29.8k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  29.8k|    do { \
  |  |  |  | 3921|  25.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  25.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 25.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 25.5k, False: 4.26k]
  |  |  ------------------
  ------------------
 3989|  4.26k|    update_cdf_2d(2, 1, m.globalmv_mode);
  ------------------
  |  | 3924|  12.7k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  12.7k|    do { \
  |  |  |  | 3921|  8.52k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  8.52k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 8.52k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 8.52k, False: 4.26k]
  |  |  ------------------
  ------------------
 3990|  4.26k|    update_cdf_2d(6, 1, m.refmv_mode);
  ------------------
  |  | 3924|  29.8k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  29.8k|    do { \
  |  |  |  | 3921|  25.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  25.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 25.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 25.5k, False: 4.26k]
  |  |  ------------------
  ------------------
 3991|  4.26k|    update_cdf_2d(3, 1, m.drl_bit);
  ------------------
  |  | 3924|  17.0k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  17.0k|    do { \
  |  |  |  | 3921|  12.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  12.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 12.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 12.7k, False: 4.26k]
  |  |  ------------------
  ------------------
 3992|  4.26k|    update_cdf_2d(4, 1, m.intra);
  ------------------
  |  | 3924|  21.3k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  21.3k|    do { \
  |  |  |  | 3921|  17.0k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  17.0k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 17.0k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 17.0k, False: 4.26k]
  |  |  ------------------
  ------------------
 3993|  4.26k|    update_cdf_2d(5, 1, m.comp);
  ------------------
  |  | 3924|  25.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  25.5k|    do { \
  |  |  |  | 3921|  21.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  21.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 21.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 21.3k, False: 4.26k]
  |  |  ------------------
  ------------------
 3994|  4.26k|    update_cdf_2d(5, 1, m.comp_dir);
  ------------------
  |  | 3924|  25.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  25.5k|    do { \
  |  |  |  | 3921|  21.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  21.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 21.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 21.3k, False: 4.26k]
  |  |  ------------------
  ------------------
 3995|  4.26k|    update_cdf_2d(6, 1, m.jnt_comp);
  ------------------
  |  | 3924|  29.8k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  29.8k|    do { \
  |  |  |  | 3921|  25.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  25.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 25.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 25.5k, False: 4.26k]
  |  |  ------------------
  ------------------
 3996|  4.26k|    update_cdf_2d(6, 1, m.mask_comp);
  ------------------
  |  | 3924|  29.8k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  29.8k|    do { \
  |  |  |  | 3921|  25.5k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  25.5k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 25.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 25.5k, False: 4.26k]
  |  |  ------------------
  ------------------
 3997|  4.26k|    update_cdf_2d(9, 1, m.wedge_comp);
  ------------------
  |  | 3924|  42.6k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  42.6k|    do { \
  |  |  |  | 3921|  38.3k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  38.3k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 38.3k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 38.3k, False: 4.26k]
  |  |  ------------------
  ------------------
 3998|  4.26k|    update_cdf_3d(6, 3, 1, m.ref);
  ------------------
  |  | 3926|  29.8k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|   102k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  80.9k|    do { \
  |  |  |  |  |  | 3921|  76.7k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  76.7k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 76.7k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 76.7k, False: 25.5k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 25.5k, False: 4.26k]
  |  |  ------------------
  ------------------
 3999|  4.26k|    update_cdf_3d(3, 3, 1, m.comp_fwd_ref);
  ------------------
  |  | 3926|  17.0k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  51.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  42.6k|    do { \
  |  |  |  |  |  | 3921|  38.3k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  38.3k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 38.3k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 38.3k, False: 12.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 12.7k, False: 4.26k]
  |  |  ------------------
  ------------------
 4000|  4.26k|    update_cdf_3d(2, 3, 1, m.comp_bwd_ref);
  ------------------
  |  | 3926|  12.7k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  34.0k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  29.8k|    do { \
  |  |  |  |  |  | 3921|  25.5k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  25.5k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 25.5k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 25.5k, False: 8.52k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 8.52k, False: 4.26k]
  |  |  ------------------
  ------------------
 4001|  4.26k|    update_cdf_3d(3, 3, 1, m.comp_uni_ref);
  ------------------
  |  | 3926|  17.0k|    for (int k = 0; k < (n1d); k++) update_cdf_2d(n2d, n3d, name[k])
  |  |  ------------------
  |  |  |  | 3924|  51.1k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  |  |  ------------------
  |  |  |  |  |  | 3920|  42.6k|    do { \
  |  |  |  |  |  | 3921|  38.3k|        dst->name[n1d] = 0; \
  |  |  |  |  |  | 3922|  38.3k|    } while (0)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (3922:14): [Folded, False: 38.3k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3924:21): [True: 38.3k, False: 12.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3926:21): [True: 12.7k, False: 4.26k]
  |  |  ------------------
  ------------------
 4002|  4.26k|    update_cdf_2d(3, 1, m.seg_pred);
  ------------------
  |  | 3924|  17.0k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  17.0k|    do { \
  |  |  |  | 3921|  12.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  12.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 12.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 12.7k, False: 4.26k]
  |  |  ------------------
  ------------------
 4003|  4.26k|    update_cdf_2d(4, 1, m.interintra);
  ------------------
  |  | 3924|  21.3k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  21.3k|    do { \
  |  |  |  | 3921|  17.0k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  17.0k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 17.0k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 17.0k, False: 4.26k]
  |  |  ------------------
  ------------------
 4004|  4.26k|    update_cdf_2d(7, 1, m.interintra_wedge);
  ------------------
  |  | 3924|  34.0k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  34.0k|    do { \
  |  |  |  | 3921|  29.8k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  29.8k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 29.8k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 29.8k, False: 4.26k]
  |  |  ------------------
  ------------------
 4005|  4.26k|    update_cdf_2d(N_BS_SIZES, 1, m.obmc);
  ------------------
  |  | 3924|  98.0k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  98.0k|    do { \
  |  |  |  | 3921|  93.7k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  93.7k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 93.7k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 93.7k, False: 4.26k]
  |  |  ------------------
  ------------------
 4006|       |
 4007|  12.7k|    for (int k = 0; k < 2; k++) {
  ------------------
  |  Branch (4007:21): [True: 8.52k, False: 4.26k]
  ------------------
 4008|  8.52k|        update_cdf_1d(10, mv.comp[k].classes);
  ------------------
  |  | 3920|  8.52k|    do { \
  |  | 3921|  8.52k|        dst->name[n1d] = 0; \
  |  | 3922|  8.52k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 8.52k]
  |  |  ------------------
  ------------------
 4009|  8.52k|        update_cdf_1d(1, mv.comp[k].sign);
  ------------------
  |  | 3920|  8.52k|    do { \
  |  | 3921|  8.52k|        dst->name[n1d] = 0; \
  |  | 3922|  8.52k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 8.52k]
  |  |  ------------------
  ------------------
 4010|  8.52k|        update_cdf_1d(1, mv.comp[k].class0);
  ------------------
  |  | 3920|  8.52k|    do { \
  |  | 3921|  8.52k|        dst->name[n1d] = 0; \
  |  | 3922|  8.52k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 8.52k]
  |  |  ------------------
  ------------------
 4011|  8.52k|        update_cdf_2d(2, 3, mv.comp[k].class0_fp);
  ------------------
  |  | 3924|  25.5k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  17.0k|    do { \
  |  |  |  | 3921|  17.0k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  17.0k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 17.0k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 17.0k, False: 8.52k]
  |  |  ------------------
  ------------------
 4012|  8.52k|        update_cdf_1d(1, mv.comp[k].class0_hp);
  ------------------
  |  | 3920|  8.52k|    do { \
  |  | 3921|  8.52k|        dst->name[n1d] = 0; \
  |  | 3922|  8.52k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 8.52k]
  |  |  ------------------
  ------------------
 4013|  8.52k|        update_cdf_2d(10, 1, mv.comp[k].classN);
  ------------------
  |  | 3924|  93.7k|    for (int j = 0; j < (n1d); j++) update_cdf_1d(n2d, name[j])
  |  |  ------------------
  |  |  |  | 3920|  85.2k|    do { \
  |  |  |  | 3921|  85.2k|        dst->name[n1d] = 0; \
  |  |  |  | 3922|  85.2k|    } while (0)
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (3922:14): [Folded, False: 85.2k]
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (3924:21): [True: 85.2k, False: 8.52k]
  |  |  ------------------
  ------------------
 4014|  8.52k|        update_cdf_1d(3, mv.comp[k].classN_fp);
  ------------------
  |  | 3920|  8.52k|    do { \
  |  | 3921|  8.52k|        dst->name[n1d] = 0; \
  |  | 3922|  8.52k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 8.52k]
  |  |  ------------------
  ------------------
 4015|  8.52k|        update_cdf_1d(1, mv.comp[k].classN_hp);
  ------------------
  |  | 3920|  8.52k|    do { \
  |  | 3921|  8.52k|        dst->name[n1d] = 0; \
  |  | 3922|  8.52k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 8.52k]
  |  |  ------------------
  ------------------
 4016|  8.52k|    }
 4017|  4.26k|    update_cdf_1d(N_MV_JOINTS - 1, mv.joint);
  ------------------
  |  | 3920|  4.26k|    do { \
  |  | 3921|  4.26k|        dst->name[n1d] = 0; \
  |  | 3922|  4.26k|    } while (0)
  |  |  ------------------
  |  |  |  Branch (3922:14): [Folded, False: 4.26k]
  |  |  ------------------
  ------------------
 4018|  4.26k|}
dav1d_cdf_thread_init_static:
 4023|   278k|void dav1d_cdf_thread_init_static(CdfThreadContext *const cdf, const unsigned qidx) {
 4024|       |    cdf->ref = NULL;
 4025|   278k|    cdf->data.qcat = (qidx > 20) + (qidx > 60) + (qidx > 120);
 4026|   278k|}
dav1d_cdf_thread_copy:
 4028|   336k|void dav1d_cdf_thread_copy(CdfContext *const dst, const CdfThreadContext *const src) {
 4029|   336k|    if (src->ref) {
  ------------------
  |  Branch (4029:9): [True: 33.8k, False: 302k]
  ------------------
 4030|  33.8k|        memcpy(dst, src->data.cdf, sizeof(*dst));
 4031|   302k|    } else {
 4032|   302k|        dst->coef = default_coef_cdf[src->data.qcat];
 4033|   302k|        memcpy(&dst->m, &default_cdf.m,
 4034|   302k|               offsetof(CdfDefaultContext, mv.joint));
 4035|   302k|        memcpy(&dst->mv.comp[1], &default_cdf.mv.comp,
 4036|       |               sizeof(default_cdf) - offsetof(CdfDefaultContext, mv.comp));
 4037|   302k|    }
 4038|   336k|}
dav1d_cdf_thread_alloc:
 4042|  55.3k|{
 4043|  55.3k|    cdf->ref = dav1d_ref_create_using_pool(c->cdf_pool,
 4044|  55.3k|                                           sizeof(CdfContext) + sizeof(atomic_uint));
 4045|  55.3k|    if (!cdf->ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (4045:9): [True: 0, False: 55.3k]
  ------------------
 4046|  55.3k|    cdf->data.cdf = cdf->ref->data;
 4047|  55.3k|    if (have_frame_mt) {
  ------------------
  |  Branch (4047:9): [True: 55.3k, False: 0]
  ------------------
 4048|  55.3k|        cdf->progress = (atomic_uint *) &cdf->data.cdf[1];
 4049|       |        atomic_init(cdf->progress, 0);
 4050|  55.3k|    }
 4051|  55.3k|    return 0;
 4052|  55.3k|}
dav1d_cdf_thread_ref:
 4056|  2.57M|{
 4057|  2.57M|    *dst = *src;
 4058|  2.57M|    if (src->ref)
  ------------------
  |  Branch (4058:9): [True: 463k, False: 2.11M]
  ------------------
 4059|   463k|        dav1d_ref_inc(src->ref);
 4060|  2.57M|}
dav1d_cdf_thread_unref:
 4062|  3.16M|void dav1d_cdf_thread_unref(CdfThreadContext *const cdf) {
 4063|       |    memset(&cdf->data, 0, sizeof(*cdf) - offsetof(CdfThreadContext, data));
 4064|  3.16M|    dav1d_ref_dec(&cdf->ref);
 4065|  3.16M|}

dav1d_init_cpu:
   63|      1|COLD void dav1d_init_cpu(void) {
   64|      1|#if HAVE_ASM && !__has_feature(memory_sanitizer)
   65|       |// memory sanitizer is inherently incompatible with asm
   66|       |#if ARCH_AARCH64 || ARCH_ARM
   67|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_arm();
   68|       |#elif ARCH_LOONGARCH
   69|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_loongarch();
   70|       |#elif ARCH_PPC64LE
   71|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_ppc();
   72|       |#elif ARCH_RISCV
   73|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_riscv();
   74|       |#elif ARCH_X86
   75|       |    dav1d_cpu_flags = dav1d_get_cpu_flags_x86();
   76|      1|#endif
   77|      1|#endif
   78|      1|}

cpu.c:dav1d_get_default_cpu_flags:
   58|      1|static ALWAYS_INLINE unsigned dav1d_get_default_cpu_flags(void) {
   59|      1|    unsigned flags = 0;
   60|       |
   61|       |#if ARCH_AARCH64 || ARCH_ARM
   62|       |#if defined(__ARM_NEON) || defined(__APPLE__) || defined(_WIN32) || ARCH_AARCH64
   63|       |    flags |= DAV1D_ARM_CPU_FLAG_NEON;
   64|       |#endif
   65|       |#ifdef __ARM_FEATURE_DOTPROD
   66|       |    flags |= DAV1D_ARM_CPU_FLAG_DOTPROD;
   67|       |#endif
   68|       |#ifdef __ARM_FEATURE_MATMUL_INT8
   69|       |    flags |= DAV1D_ARM_CPU_FLAG_I8MM;
   70|       |#endif
   71|       |#if ARCH_AARCH64
   72|       |#ifdef __ARM_FEATURE_SVE
   73|       |    flags |= DAV1D_ARM_CPU_FLAG_SVE;
   74|       |#endif
   75|       |#ifdef __ARM_FEATURE_SVE2
   76|       |    flags |= DAV1D_ARM_CPU_FLAG_SVE2;
   77|       |#endif
   78|       |#endif /* ARCH_AARCH64 */
   79|       |#elif ARCH_PPC64LE
   80|       |#if defined(__VSX__)
   81|       |    flags |= DAV1D_PPC_CPU_FLAG_VSX;
   82|       |#endif
   83|       |#if defined(__POWER9_VECTOR__)
   84|       |    flags |= DAV1D_PPC_CPU_FLAG_PWR9;
   85|       |#endif
   86|       |#elif ARCH_RISCV
   87|       |#if defined(__riscv_v)
   88|       |    flags |= DAV1D_RISCV_CPU_FLAG_V;
   89|       |#endif
   90|       |#elif ARCH_X86
   91|       |#if defined(__AVX512F__) && defined(__AVX512CD__) && \
   92|       |    defined(__AVX512BW__) && defined(__AVX512DQ__) && \
   93|       |    defined(__AVX512VL__) && defined(__AVX512VNNI__) && \
   94|       |    defined(__AVX512IFMA__) && defined(__AVX512VBMI__) && \
   95|       |    defined(__AVX512VBMI2__) && defined(__AVX512VPOPCNTDQ__) && \
   96|       |    defined(__AVX512BITALG__) && defined(__GFNI__) && \
   97|       |    defined(__VAES__) && defined(__VPCLMULQDQ__)
   98|       |    flags |= DAV1D_X86_CPU_FLAG_AVX512ICL |
   99|       |             DAV1D_X86_CPU_FLAG_AVX2 |
  100|       |             DAV1D_X86_CPU_FLAG_SSE41 |
  101|       |             DAV1D_X86_CPU_FLAG_SSSE3 |
  102|       |             DAV1D_X86_CPU_FLAG_SSE2;
  103|       |#elif defined(__AVX2__)
  104|       |    flags |= DAV1D_X86_CPU_FLAG_AVX2 |
  105|       |             DAV1D_X86_CPU_FLAG_SSE41 |
  106|       |             DAV1D_X86_CPU_FLAG_SSSE3 |
  107|       |             DAV1D_X86_CPU_FLAG_SSE2;
  108|       |#elif defined(__SSE4_1__) || defined(__AVX__)
  109|       |    flags |= DAV1D_X86_CPU_FLAG_SSE41 |
  110|       |             DAV1D_X86_CPU_FLAG_SSSE3 |
  111|       |             DAV1D_X86_CPU_FLAG_SSE2;
  112|       |#elif defined(__SSSE3__)
  113|       |    flags |= DAV1D_X86_CPU_FLAG_SSSE3 |
  114|       |             DAV1D_X86_CPU_FLAG_SSE2;
  115|       |#elif ARCH_X86_64 || defined(__SSE2__) || \
  116|       |      (defined(_M_IX86_FP) && _M_IX86_FP >= 2)
  117|       |    flags |= DAV1D_X86_CPU_FLAG_SSE2;
  118|      1|#endif
  119|      1|#endif
  120|       |
  121|      1|    return flags;
  122|      1|}
pal.c:dav1d_get_cpu_flags:
  124|  9.41k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  9.41k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  9.41k|    return flags;
  134|  9.41k|}
refmvs.c:dav1d_get_cpu_flags:
  124|  9.41k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  9.41k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  9.41k|    return flags;
  134|  9.41k|}
msac.c:dav1d_get_cpu_flags:
  124|   315k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|   315k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|   315k|    return flags;
  134|   315k|}
cdef_tmpl.c:dav1d_get_cpu_flags:
  124|  8.57k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.57k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.57k|    return flags;
  134|  8.57k|}
filmgrain_tmpl.c:dav1d_get_cpu_flags:
  124|  8.57k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.57k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.57k|    return flags;
  134|  8.57k|}
ipred_tmpl.c:dav1d_get_cpu_flags:
  124|  8.57k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.57k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.57k|    return flags;
  134|  8.57k|}
itx_tmpl.c:dav1d_get_cpu_flags:
  124|  8.57k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.57k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.57k|    return flags;
  134|  8.57k|}
loopfilter_tmpl.c:dav1d_get_cpu_flags:
  124|  8.57k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.57k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.57k|    return flags;
  134|  8.57k|}
looprestoration_tmpl.c:dav1d_get_cpu_flags:
  124|  8.57k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.57k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.57k|    return flags;
  134|  8.57k|}
mc_tmpl.c:dav1d_get_cpu_flags:
  124|  8.57k|static ALWAYS_INLINE unsigned dav1d_get_cpu_flags(void) {
  125|  8.57k|    unsigned flags = dav1d_cpu_flags & dav1d_cpu_flags_mask;
  126|       |
  127|       |#if TRIM_DSP_FUNCTIONS
  128|       |/* Since this function is inlined, unconditionally setting a flag here will
  129|       | * enable dead code elimination in the calling function. */
  130|       |    flags |= dav1d_get_default_cpu_flags();
  131|       |#endif
  132|       |
  133|  8.57k|    return flags;
  134|  8.57k|}

ctx.c:memset_w1:
   34|  86.8M|static void memset_w1(void *const ptr, const int value) {
   35|  86.8M|    set_ctx1((uint8_t *) ptr, 0, value);
  ------------------
  |  |   56|  86.8M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  ------------------
   36|  86.8M|}
ctx.c:memset_w2:
   38|  25.7M|static void memset_w2(void *const ptr, const int value) {
   39|  25.7M|    set_ctx2((uint8_t *) ptr, 0, value);
  ------------------
  |  |   58|  25.7M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  ------------------
   40|  25.7M|}
ctx.c:memset_w4:
   42|  19.2M|static void memset_w4(void *const ptr, const int value) {
   43|  19.2M|    set_ctx4((uint8_t *) ptr, 0, value);
  ------------------
  |  |   60|  19.2M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  ------------------
   44|  19.2M|}
ctx.c:memset_w8:
   46|  14.7M|static void memset_w8(void *const ptr, const int value) {
   47|  14.7M|    set_ctx8((uint8_t *) ptr, 0, value);
  ------------------
  |  |   62|  14.7M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  ------------------
   48|  14.7M|}
ctx.c:memset_w16:
   50|  15.8M|static void memset_w16(void *const ptr, const int value) {
   51|  15.8M|    set_ctx16((uint8_t *) ptr, 0, value);
  ------------------
  |  |   63|  15.8M|#define set_ctx16(var, off, val) do { \
  |  |   64|  15.8M|        memset(&(var)[off], val, 16); \
  |  |   65|  15.8M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (65:14): [Folded, False: 15.8M]
  |  |  ------------------
  ------------------
   52|  15.8M|}
ctx.c:memset_w32:
   54|  4.03M|static void memset_w32(void *const ptr, const int value) {
   55|  4.03M|    set_ctx32((uint8_t *) ptr, 0, value);
  ------------------
  |  |   66|  4.03M|#define set_ctx32(var, off, val) do { \
  |  |   67|  4.03M|        memset(&(var)[off], val, 32); \
  |  |   68|  4.03M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (68:14): [Folded, False: 4.03M]
  |  |  ------------------
  ------------------
   56|  4.03M|}

lf_mask.c:dav1d_memset_likely_pow2:
   44|  7.37M|static inline void dav1d_memset_likely_pow2(void *const ptr, const int value, const int n) {
   45|  7.37M|    assert(n >= 1 && n <= 32);
  ------------------
  |  Branch (45:5): [True: 7.37M, False: 812]
  |  Branch (45:5): [True: 7.37M, False: 366]
  ------------------
   46|  7.37M|    if ((n&(n-1)) == 0) {
  ------------------
  |  Branch (46:9): [True: 6.75M, False: 623k]
  ------------------
   47|  6.75M|        dav1d_memset_pow2[ulog2(n)](ptr, value);
   48|  6.75M|    } else {
   49|   623k|        memset(ptr, value, n);
   50|   623k|    }
   51|  7.37M|}
recon_tmpl.c:dav1d_memset_likely_pow2:
   44|  99.1M|static inline void dav1d_memset_likely_pow2(void *const ptr, const int value, const int n) {
   45|  99.1M|    assert(n >= 1 && n <= 32);
  ------------------
  |  Branch (45:5): [True: 99.0M, False: 13.6k]
  |  Branch (45:5): [True: 99.1M, False: 18.4E]
  ------------------
   46|  99.1M|    if ((n&(n-1)) == 0) {
  ------------------
  |  Branch (46:9): [True: 97.5M, False: 1.56M]
  ------------------
   47|  97.5M|        dav1d_memset_pow2[ulog2(n)](ptr, value);
   48|  97.5M|    } else {
   49|  1.56M|        memset(ptr, value, n);
   50|  1.56M|    }
   51|  99.1M|}

dav1d_data_create_internal:
   43|   411k|uint8_t *dav1d_data_create_internal(Dav1dData *const buf, const size_t sz) {
   44|   411k|    validate_input_or_ret(buf != NULL, NULL);
  ------------------
  |  |   52|   411k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 411k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
   45|       |
   46|   411k|    if (sz > SIZE_MAX / 2) return NULL;
  ------------------
  |  Branch (46:9): [True: 0, False: 411k]
  ------------------
   47|   411k|    buf->ref = dav1d_ref_create(ALLOC_DAV1DDATA, sz);
  ------------------
  |  |   49|   411k|#define dav1d_ref_create(type, size) dav1d_ref_create(size)
  ------------------
   48|   411k|    if (!buf->ref) return NULL;
  ------------------
  |  Branch (48:9): [True: 0, False: 411k]
  ------------------
   49|   411k|    buf->data = buf->ref->const_data;
   50|   411k|    buf->sz = sz;
   51|   411k|    dav1d_data_props_set_defaults(&buf->m);
   52|   411k|    buf->m.size = sz;
   53|       |
   54|   411k|    return buf->ref->data;
   55|   411k|}
dav1d_data_ref:
   98|   771k|void dav1d_data_ref(Dav1dData *const dst, const Dav1dData *const src) {
   99|   771k|    assert(dst != NULL);
  ------------------
  |  Branch (99:5): [True: 771k, False: 0]
  ------------------
  100|   771k|    assert(dst->data == NULL);
  ------------------
  |  Branch (100:5): [True: 771k, False: 0]
  ------------------
  101|   771k|    assert(src != NULL);
  ------------------
  |  Branch (101:5): [True: 771k, False: 0]
  ------------------
  102|       |
  103|   771k|    if (src->ref) {
  ------------------
  |  Branch (103:9): [True: 771k, False: 0]
  ------------------
  104|   771k|        assert(src->data != NULL);
  ------------------
  |  Branch (104:9): [True: 771k, False: 0]
  ------------------
  105|   771k|        dav1d_ref_inc(src->ref);
  106|   771k|    }
  107|   771k|    if (src->m.user_data.ref) dav1d_ref_inc(src->m.user_data.ref);
  ------------------
  |  Branch (107:9): [True: 0, False: 771k]
  ------------------
  108|   771k|    *dst = *src;
  109|   771k|}
dav1d_data_props_copy:
  113|   658k|{
  114|   658k|    assert(dst != NULL);
  ------------------
  |  Branch (114:5): [True: 658k, False: 0]
  ------------------
  115|   658k|    assert(src != NULL);
  ------------------
  |  Branch (115:5): [True: 658k, False: 0]
  ------------------
  116|       |
  117|   658k|    dav1d_ref_dec(&dst->user_data.ref);
  118|   658k|    *dst = *src;
  119|   658k|    if (dst->user_data.ref) dav1d_ref_inc(dst->user_data.ref);
  ------------------
  |  Branch (119:9): [True: 0, False: 658k]
  ------------------
  120|   658k|}
dav1d_data_props_set_defaults:
  122|  6.75M|void dav1d_data_props_set_defaults(Dav1dDataProps *const props) {
  123|  6.75M|    assert(props != NULL);
  ------------------
  |  Branch (123:5): [True: 6.75M, False: 9]
  ------------------
  124|       |
  125|  6.75M|    memset(props, 0, sizeof(*props));
  126|       |    props->timestamp = INT64_MIN;
  127|  6.75M|    props->offset = -1;
  128|  6.75M|}
dav1d_data_props_unref_internal:
  130|  9.41k|void dav1d_data_props_unref_internal(Dav1dDataProps *const props) {
  131|  9.41k|    validate_input(props != NULL);
  ------------------
  |  |   59|  9.41k|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|  9.41k|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 9.41k]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  132|       |
  133|  9.41k|    struct Dav1dRef *user_data_ref = props->user_data.ref;
  134|  9.41k|    dav1d_data_props_set_defaults(props);
  135|  9.41k|    dav1d_ref_dec(&user_data_ref);
  136|  9.41k|}
dav1d_data_unref_internal:
  138|  1.19M|void dav1d_data_unref_internal(Dav1dData *const buf) {
  139|  1.19M|    validate_input(buf != NULL);
  ------------------
  |  |   59|  1.19M|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|  1.19M|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 1.19M]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  140|       |
  141|  1.19M|    struct Dav1dRef *user_data_ref = buf->m.user_data.ref;
  142|  1.19M|    if (buf->ref) {
  ------------------
  |  Branch (142:9): [True: 1.18M, False: 9.40k]
  ------------------
  143|  1.18M|        validate_input(buf->data != NULL);
  ------------------
  |  |   59|  1.18M|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|  1.18M|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 1.18M]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  144|  1.18M|        dav1d_ref_dec(&buf->ref);
  145|  1.18M|    }
  146|  1.19M|    memset(buf, 0, sizeof(*buf));
  147|  1.19M|    dav1d_data_props_set_defaults(&buf->m);
  148|  1.19M|    dav1d_ref_dec(&user_data_ref);
  149|  1.19M|}

dav1d_decode_tile_sbrow:
 2597|  4.12M|int dav1d_decode_tile_sbrow(Dav1dTaskContext *const t) {
 2598|  4.12M|    const Dav1dFrameContext *const f = t->f;
 2599|  4.12M|    const enum BlockLevel root_bl = f->seq_hdr->sb128 ? BL_128X128 : BL_64X64;
  ------------------
  |  Branch (2599:37): [True: 2.10M, False: 2.02M]
  ------------------
 2600|  4.12M|    Dav1dTileState *const ts = t->ts;
 2601|  4.12M|    const Dav1dContext *const c = f->c;
 2602|  4.12M|    const int sb_step = f->sb_step;
 2603|  4.12M|    const int tile_row = ts->tiling.row, tile_col = ts->tiling.col;
 2604|  4.12M|    const int col_sb_start = f->frame_hdr->tiling.col_start_sb[tile_col];
 2605|  4.12M|    const int col_sb128_start = col_sb_start >> !f->seq_hdr->sb128;
 2606|       |
 2607|  4.12M|    if (IS_INTER_OR_SWITCH(f->frame_hdr) || f->frame_hdr->allow_intrabc) {
  ------------------
  |  |   36|  8.25M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 3.05M, False: 1.06M]
  |  |  ------------------
  ------------------
  |  Branch (2607:45): [True: 958k, False: 110k]
  ------------------
 2608|  4.01M|        dav1d_refmvs_tile_sbrow_init(&t->rt, &f->rf, ts->tiling.col_start,
 2609|  4.01M|                                     ts->tiling.col_end, ts->tiling.row_start,
 2610|  4.01M|                                     ts->tiling.row_end, t->by >> f->sb_shift,
 2611|  4.01M|                                     ts->tiling.row, t->frame_thread.pass);
 2612|  4.01M|    }
 2613|       |
 2614|  4.12M|    if (IS_INTER_OR_SWITCH(f->frame_hdr) && c->n_fc > 1) {
  ------------------
  |  |   36|  8.25M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 3.05M, False: 1.07M]
  |  |  ------------------
  ------------------
  |  Branch (2614:45): [True: 3.05M, False: 18.4E]
  ------------------
 2615|  3.05M|        const int sby = (t->by - ts->tiling.row_start) >> f->sb_shift;
 2616|  3.05M|        int (*const lowest_px)[2] = ts->lowest_pixel[sby];
 2617|  24.4M|        for (int n = 0; n < 7; n++)
  ------------------
  |  Branch (2617:25): [True: 21.3M, False: 3.05M]
  ------------------
 2618|  64.1M|            for (int m = 0; m < 2; m++)
  ------------------
  |  Branch (2618:29): [True: 42.7M, False: 21.3M]
  ------------------
 2619|  42.7M|                lowest_px[n][m] = INT_MIN;
 2620|  3.05M|    }
 2621|       |
 2622|  4.12M|    reset_context(&t->l, IS_KEY_OR_INTRA(f->frame_hdr), t->frame_thread.pass);
  ------------------
  |  |   43|  4.12M|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  4.12M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  ------------------
 2623|  4.12M|    if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (2623:9): [True: 1.98M, False: 2.13M]
  ------------------
 2624|  18.4E|        const int off_2pass = c->n_tc > 1 ? f->sb128w * f->frame_hdr->tiling.rows : 0;
  ------------------
  |  Branch (2624:31): [True: 1.98M, False: 18.4E]
  ------------------
 2625|  1.98M|        for (t->bx = ts->tiling.col_start,
 2626|  1.98M|             t->a = f->a + off_2pass + col_sb128_start + tile_row * f->sb128w;
 2627|  4.41M|             t->bx < ts->tiling.col_end; t->bx += sb_step)
  ------------------
  |  Branch (2627:14): [True: 2.42M, False: 1.98M]
  ------------------
 2628|  2.42M|        {
 2629|  2.42M|            if (atomic_load_explicit(c->flush, memory_order_acquire))
  ------------------
  |  Branch (2629:17): [True: 7, False: 2.42M]
  ------------------
 2630|      7|                return 1;
 2631|  2.42M|            if (decode_sb(t, root_bl, dav1d_intra_edge_tree[root_bl]))
  ------------------
  |  Branch (2631:17): [True: 0, False: 2.42M]
  ------------------
 2632|      0|                return 1;
 2633|  2.42M|            if (t->bx & 16 || f->seq_hdr->sb128)
  ------------------
  |  Branch (2633:17): [True: 653k, False: 1.77M]
  |  Branch (2633:31): [True: 1.09M, False: 675k]
  ------------------
 2634|  1.74M|                t->a++;
 2635|  2.42M|        }
 2636|  1.98M|        f->bd_fn.backup_ipred_edge(t);
 2637|  1.98M|        return 0;
 2638|  1.98M|    }
 2639|       |
 2640|  2.13M|    if (f->c->n_tc > 1 && f->frame_hdr->use_ref_frame_mvs) {
  ------------------
  |  Branch (2640:9): [True: 2.13M, False: 576]
  |  Branch (2640:27): [True: 827k, False: 1.30M]
  ------------------
 2641|   827k|        f->c->refmvs_dsp.load_tmvs(&f->rf, ts->tiling.row,
 2642|   827k|                                   ts->tiling.col_start >> 1, ts->tiling.col_end >> 1,
 2643|   827k|                                   t->by >> 1, (t->by + sb_step) >> 1);
 2644|   827k|    }
 2645|  2.13M|    memset(t->pal_sz_uv[1], 0, sizeof(*t->pal_sz_uv));
 2646|  2.13M|    const int sb128y = t->by >> 5;
 2647|  2.13M|    for (t->bx = ts->tiling.col_start, t->a = f->a + col_sb128_start + tile_row * f->sb128w,
 2648|  2.13M|         t->lf_mask = f->lf.mask + sb128y * f->sb128w + col_sb128_start;
 2649|  5.23M|         t->bx < ts->tiling.col_end; t->bx += sb_step)
  ------------------
  |  Branch (2649:10): [True: 3.17M, False: 2.06M]
  ------------------
 2650|  3.17M|    {
 2651|  3.17M|        if (atomic_load_explicit(c->flush, memory_order_acquire))
  ------------------
  |  Branch (2651:13): [True: 377, False: 3.17M]
  ------------------
 2652|    377|            return 1;
 2653|  3.17M|        if (root_bl == BL_128X128) {
  ------------------
  |  Branch (2653:13): [True: 1.53M, False: 1.63M]
  ------------------
 2654|  1.53M|            t->cur_sb_cdef_idx_ptr = t->lf_mask->cdef_idx;
 2655|  1.53M|            t->cur_sb_cdef_idx_ptr[0] = -1;
 2656|  1.53M|            t->cur_sb_cdef_idx_ptr[1] = -1;
 2657|  1.53M|            t->cur_sb_cdef_idx_ptr[2] = -1;
 2658|  1.53M|            t->cur_sb_cdef_idx_ptr[3] = -1;
 2659|  1.63M|        } else {
 2660|  1.63M|            t->cur_sb_cdef_idx_ptr =
 2661|  1.63M|                &t->lf_mask->cdef_idx[((t->bx & 16) >> 4) +
 2662|  1.63M|                                      ((t->by & 16) >> 3)];
 2663|  1.63M|            t->cur_sb_cdef_idx_ptr[0] = -1;
 2664|  1.63M|        }
 2665|       |        // Restoration filter
 2666|  12.6M|        for (int p = 0; p < 3; p++) {
  ------------------
  |  Branch (2666:25): [True: 9.50M, False: 3.17M]
  ------------------
 2667|  9.50M|            if (!((f->lf.restore_planes >> p) & 1U))
  ------------------
  |  Branch (2667:17): [True: 8.80M, False: 694k]
  ------------------
 2668|  8.80M|                continue;
 2669|       |
 2670|   694k|            const int ss_ver = p && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (2670:32): [True: 206k, False: 488k]
  |  Branch (2670:37): [True: 150k, False: 55.7k]
  ------------------
 2671|   694k|            const int ss_hor = p && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (2671:32): [True: 206k, False: 488k]
  |  Branch (2671:37): [True: 162k, False: 43.9k]
  ------------------
 2672|   694k|            const int unit_size_log2 = f->frame_hdr->restoration.unit_size[!!p];
 2673|   694k|            const int y = t->by * 4 >> ss_ver;
 2674|   694k|            const int h = (f->cur.p.h + ss_ver) >> ss_ver;
 2675|       |
 2676|   694k|            const int unit_size = 1 << unit_size_log2;
 2677|   694k|            const unsigned mask = unit_size - 1;
 2678|   694k|            if (y & mask) continue;
  ------------------
  |  Branch (2678:17): [True: 248k, False: 446k]
  ------------------
 2679|   446k|            const int half_unit = unit_size >> 1;
 2680|       |            // Round half up at frame boundaries, if there's more than one
 2681|       |            // restoration unit
 2682|   446k|            if (y && y + half_unit > h) continue;
  ------------------
  |  Branch (2682:17): [True: 252k, False: 193k]
  |  Branch (2682:22): [True: 11.9k, False: 240k]
  ------------------
 2683|       |
 2684|   434k|            const enum Dav1dRestorationType frame_type = f->frame_hdr->restoration.type[p];
 2685|       |
 2686|   434k|            if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (2686:17): [True: 45.2k, False: 389k]
  ------------------
 2687|  45.2k|                const int w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
 2688|  45.2k|                const int n_units = imax(1, (w + half_unit) >> unit_size_log2);
 2689|       |
 2690|  45.2k|                const int d = f->frame_hdr->super_res.width_scale_denominator;
 2691|  45.2k|                const int rnd = unit_size * 8 - 1, shift = unit_size_log2 + 3;
 2692|  45.2k|                const int x0 = ((4 *  t->bx            * d >> ss_hor) + rnd) >> shift;
 2693|  45.2k|                const int x1 = ((4 * (t->bx + sb_step) * d >> ss_hor) + rnd) >> shift;
 2694|       |
 2695|  88.5k|                for (int x = x0; x < imin(x1, n_units); x++) {
  ------------------
  |  Branch (2695:34): [True: 43.2k, False: 45.2k]
  ------------------
 2696|  43.2k|                    const int px_x = x << (unit_size_log2 + ss_hor);
 2697|  43.2k|                    const int sb_idx = (t->by >> 5) * f->sr_sb128w + (px_x >> 7);
 2698|  43.2k|                    const int unit_idx = ((t->by & 16) >> 3) + ((px_x & 64) >> 6);
 2699|  43.2k|                    Av1RestorationUnit *const lr = &f->lf.lr_mask[sb_idx].lr[p][unit_idx];
 2700|       |
 2701|  43.2k|                    read_restoration_info(t, lr, p, frame_type);
 2702|  43.2k|                }
 2703|   389k|            } else {
 2704|   389k|                const int x = 4 * t->bx >> ss_hor;
 2705|   389k|                if (x & mask) continue;
  ------------------
  |  Branch (2705:21): [True: 171k, False: 217k]
  ------------------
 2706|   217k|                const int w = (f->cur.p.w + ss_hor) >> ss_hor;
 2707|       |                // Round half up at frame boundaries, if there's more than one
 2708|       |                // restoration unit
 2709|   217k|                if (x && x + half_unit > w) continue;
  ------------------
  |  Branch (2709:21): [True: 99.1k, False: 118k]
  |  Branch (2709:26): [True: 1.87k, False: 97.3k]
  ------------------
 2710|   215k|                const int sb_idx = (t->by >> 5) * f->sr_sb128w + (t->bx >> 5);
 2711|   215k|                const int unit_idx = ((t->by & 16) >> 3) + ((t->bx & 16) >> 4);
 2712|   215k|                Av1RestorationUnit *const lr = &f->lf.lr_mask[sb_idx].lr[p][unit_idx];
 2713|       |
 2714|   215k|                read_restoration_info(t, lr, p, frame_type);
 2715|   215k|            }
 2716|   434k|        }
 2717|  3.17M|        if (decode_sb(t, root_bl, dav1d_intra_edge_tree[root_bl]))
  ------------------
  |  Branch (2717:13): [True: 70.9k, False: 3.10M]
  ------------------
 2718|  70.9k|            return 1;
 2719|  3.10M|        if (t->bx & 16 || f->seq_hdr->sb128) {
  ------------------
  |  Branch (2719:13): [True: 770k, False: 2.32M]
  |  Branch (2719:27): [True: 1.52M, False: 805k]
  ------------------
 2720|  2.29M|            t->a++;
 2721|  2.29M|            t->lf_mask++;
 2722|  2.29M|        }
 2723|  3.10M|    }
 2724|       |
 2725|  2.06M|    if (f->seq_hdr->ref_frame_mvs && f->c->n_tc > 1 && IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|   975k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 832k, False: 142k]
  |  |  ------------------
  ------------------
  |  Branch (2725:9): [True: 975k, False: 1.09M]
  |  Branch (2725:38): [True: 975k, False: 18.4E]
  ------------------
 2726|   832k|        dav1d_refmvs_save_tmvs(&f->c->refmvs_dsp, &t->rt,
 2727|   832k|                               ts->tiling.col_start >> 1, ts->tiling.col_end >> 1,
 2728|   832k|                               t->by >> 1, (t->by + sb_step) >> 1);
 2729|   832k|    }
 2730|       |
 2731|       |    // backup pre-loopfilter pixels for intra prediction of the next sbrow
 2732|  2.06M|    if (t->frame_thread.pass != 1)
  ------------------
  |  Branch (2732:9): [True: 0, False: 2.06M]
  ------------------
 2733|      0|        f->bd_fn.backup_ipred_edge(t);
 2734|       |
 2735|       |    // backup t->a/l.tx_lpf_y/uv at tile boundaries to use them to "fix"
 2736|       |    // up the initial value in neighbour tiles when running the loopfilter
 2737|  2.06M|    int align_h = (f->bh + 31) & ~31;
 2738|  2.06M|    memcpy(&f->lf.tx_lpf_right_edge[0][align_h * tile_col + t->by],
 2739|  2.06M|           &t->l.tx_lpf_y[t->by & 16], sb_step);
 2740|  2.06M|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2741|  2.06M|    align_h >>= ss_ver;
 2742|  2.06M|    memcpy(&f->lf.tx_lpf_right_edge[1][align_h * tile_col + (t->by >> ss_ver)],
 2743|  2.06M|           &t->l.tx_lpf_uv[(t->by & 16) >> ss_ver], sb_step >> ss_ver);
 2744|       |
 2745|       |    // error out on symbol decoder overread
 2746|  2.06M|    if (ts->msac.cnt <= -15) return 1;
  ------------------
  |  Branch (2746:9): [True: 38.1k, False: 2.02M]
  ------------------
 2747|       |
 2748|  2.02M|    return c->strict_std_compliance &&
  ------------------
  |  Branch (2748:12): [True: 0, False: 2.02M]
  ------------------
 2749|      0|           (t->by >> f->sb_shift) + 1 >= f->frame_hdr->tiling.row_start_sb[tile_row + 1] &&
  ------------------
  |  Branch (2749:12): [True: 0, False: 0]
  ------------------
 2750|      0|           check_trailing_bits_after_symbol_coder(&ts->msac);
  ------------------
  |  Branch (2750:12): [True: 0, False: 0]
  ------------------
 2751|  2.06M|}
dav1d_decode_frame_init:
 2753|   355k|int dav1d_decode_frame_init(Dav1dFrameContext *const f) {
 2754|   355k|    const Dav1dContext *const c = f->c;
 2755|   355k|    int retval = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|   355k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 2756|       |
 2757|   355k|    if (f->sbh > f->lf.start_of_tile_row_sz) {
  ------------------
  |  Branch (2757:9): [True: 17.0k, False: 338k]
  ------------------
 2758|  17.0k|        dav1d_free(f->lf.start_of_tile_row);
  ------------------
  |  |  135|  17.0k|#define dav1d_free(ptr) free(ptr)
  ------------------
 2759|  17.0k|        f->lf.start_of_tile_row = dav1d_malloc(ALLOC_TILE, f->sbh * sizeof(uint8_t));
  ------------------
  |  |  132|  17.0k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 2760|  17.0k|        if (!f->lf.start_of_tile_row) {
  ------------------
  |  Branch (2760:13): [True: 0, False: 17.0k]
  ------------------
 2761|      0|            f->lf.start_of_tile_row_sz = 0;
 2762|      0|            goto error;
 2763|      0|        }
 2764|  17.0k|        f->lf.start_of_tile_row_sz = f->sbh;
 2765|  17.0k|    }
 2766|   355k|    int sby = 0;
 2767|   780k|    for (int tile_row = 0; tile_row < f->frame_hdr->tiling.rows; tile_row++) {
  ------------------
  |  Branch (2767:28): [True: 424k, False: 355k]
  ------------------
 2768|   424k|        f->lf.start_of_tile_row[sby++] = tile_row;
 2769|  10.9M|        while (sby < f->frame_hdr->tiling.row_start_sb[tile_row + 1])
  ------------------
  |  Branch (2769:16): [True: 10.5M, False: 424k]
  ------------------
 2770|  10.5M|            f->lf.start_of_tile_row[sby++] = 0;
 2771|   424k|    }
 2772|       |
 2773|   355k|    const int n_ts = f->frame_hdr->tiling.cols * f->frame_hdr->tiling.rows;
 2774|   355k|    if (n_ts != f->n_ts) {
  ------------------
  |  Branch (2774:9): [True: 43.1k, False: 312k]
  ------------------
 2775|  43.1k|        if (c->n_fc > 1) {
  ------------------
  |  Branch (2775:13): [True: 43.1k, False: 18.4E]
  ------------------
 2776|  43.1k|            dav1d_free(f->frame_thread.tile_start_off);
  ------------------
  |  |  135|  43.1k|#define dav1d_free(ptr) free(ptr)
  ------------------
 2777|  43.1k|            f->frame_thread.tile_start_off =
 2778|  43.1k|                dav1d_malloc(ALLOC_TILE, sizeof(*f->frame_thread.tile_start_off) * n_ts);
  ------------------
  |  |  132|  43.1k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 2779|  43.1k|            if (!f->frame_thread.tile_start_off) {
  ------------------
  |  Branch (2779:17): [True: 0, False: 43.1k]
  ------------------
 2780|      0|                f->n_ts = 0;
 2781|      0|                goto error;
 2782|      0|            }
 2783|  43.1k|        }
 2784|  43.1k|        dav1d_free_aligned(f->ts);
  ------------------
  |  |  136|  43.1k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2785|  43.1k|        f->ts = dav1d_alloc_aligned(ALLOC_TILE, sizeof(*f->ts) * n_ts, 32);
  ------------------
  |  |  134|  43.1k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2786|  43.1k|        if (!f->ts) goto error;
  ------------------
  |  Branch (2786:13): [True: 0, False: 43.1k]
  ------------------
 2787|  43.1k|        f->n_ts = n_ts;
 2788|  43.1k|    }
 2789|       |
 2790|   355k|    const int a_sz = f->sb128w * f->frame_hdr->tiling.rows * (1 + (c->n_fc > 1 && c->n_tc > 1));
  ------------------
  |  Branch (2790:68): [True: 355k, False: 143]
  |  Branch (2790:83): [True: 355k, False: 18.4E]
  ------------------
 2791|   355k|    if (a_sz != f->a_sz) {
  ------------------
  |  Branch (2791:9): [True: 43.5k, False: 311k]
  ------------------
 2792|  43.5k|        dav1d_free(f->a);
  ------------------
  |  |  135|  43.5k|#define dav1d_free(ptr) free(ptr)
  ------------------
 2793|  43.5k|        f->a = dav1d_malloc(ALLOC_TILE, sizeof(*f->a) * a_sz);
  ------------------
  |  |  132|  43.5k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 2794|  43.5k|        if (!f->a) {
  ------------------
  |  Branch (2794:13): [True: 0, False: 43.5k]
  ------------------
 2795|      0|            f->a_sz = 0;
 2796|      0|            goto error;
 2797|      0|        }
 2798|  43.5k|        f->a_sz = a_sz;
 2799|  43.5k|    }
 2800|       |
 2801|   355k|    const int num_sb128 = f->sb128w * f->sb128h;
 2802|   355k|    const uint8_t *const size_mul = ss_size_mul[f->cur.p.layout];
 2803|   355k|    const int hbd = !!f->seq_hdr->hbd;
 2804|   355k|    if (c->n_fc > 1) {
  ------------------
  |  Branch (2804:9): [True: 355k, False: 94]
  ------------------
 2805|   355k|        const unsigned sb_step4 = f->sb_step * 4;
 2806|   355k|        int tile_idx = 0;
 2807|   779k|        for (int tile_row = 0; tile_row < f->frame_hdr->tiling.rows; tile_row++) {
  ------------------
  |  Branch (2807:32): [True: 424k, False: 355k]
  ------------------
 2808|   424k|            const unsigned row_off = f->frame_hdr->tiling.row_start_sb[tile_row] *
 2809|   424k|                                     sb_step4 * f->sb128w * 128;
 2810|   424k|            const unsigned b_diff = (f->frame_hdr->tiling.row_start_sb[tile_row + 1] -
 2811|   424k|                                     f->frame_hdr->tiling.row_start_sb[tile_row]) * sb_step4;
 2812|   877k|            for (int tile_col = 0; tile_col < f->frame_hdr->tiling.cols; tile_col++) {
  ------------------
  |  Branch (2812:36): [True: 453k, False: 424k]
  ------------------
 2813|   453k|                f->frame_thread.tile_start_off[tile_idx++] = row_off + b_diff *
 2814|   453k|                    f->frame_hdr->tiling.col_start_sb[tile_col] * sb_step4;
 2815|   453k|            }
 2816|   424k|        }
 2817|       |
 2818|   355k|        const int lowest_pixel_mem_sz = f->frame_hdr->tiling.cols * f->sbh;
 2819|   355k|        if (lowest_pixel_mem_sz != f->tile_thread.lowest_pixel_mem_sz) {
  ------------------
  |  Branch (2819:13): [True: 29.4k, False: 325k]
  ------------------
 2820|  29.4k|            dav1d_free(f->tile_thread.lowest_pixel_mem);
  ------------------
  |  |  135|  29.4k|#define dav1d_free(ptr) free(ptr)
  ------------------
 2821|  29.4k|            f->tile_thread.lowest_pixel_mem =
 2822|  29.4k|                dav1d_malloc(ALLOC_TILE, lowest_pixel_mem_sz *
  ------------------
  |  |  132|  29.4k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 2823|  29.4k|                             sizeof(*f->tile_thread.lowest_pixel_mem));
 2824|  29.4k|            if (!f->tile_thread.lowest_pixel_mem) {
  ------------------
  |  Branch (2824:17): [True: 0, False: 29.4k]
  ------------------
 2825|      0|                f->tile_thread.lowest_pixel_mem_sz = 0;
 2826|      0|                goto error;
 2827|      0|            }
 2828|  29.4k|            f->tile_thread.lowest_pixel_mem_sz = lowest_pixel_mem_sz;
 2829|  29.4k|        }
 2830|   355k|        int (*lowest_pixel_ptr)[7][2] = f->tile_thread.lowest_pixel_mem;
 2831|   779k|        for (int tile_row = 0, tile_row_base = 0; tile_row < f->frame_hdr->tiling.rows;
  ------------------
  |  Branch (2831:51): [True: 424k, False: 355k]
  ------------------
 2832|   424k|             tile_row++, tile_row_base += f->frame_hdr->tiling.cols)
 2833|   424k|        {
 2834|   424k|            const int tile_row_sb_h = f->frame_hdr->tiling.row_start_sb[tile_row + 1] -
 2835|   424k|                                      f->frame_hdr->tiling.row_start_sb[tile_row];
 2836|   877k|            for (int tile_col = 0; tile_col < f->frame_hdr->tiling.cols; tile_col++) {
  ------------------
  |  Branch (2836:36): [True: 453k, False: 424k]
  ------------------
 2837|   453k|                f->ts[tile_row_base + tile_col].lowest_pixel = lowest_pixel_ptr;
 2838|   453k|                lowest_pixel_ptr += tile_row_sb_h;
 2839|   453k|            }
 2840|   424k|        }
 2841|       |
 2842|   355k|        const int cbi_sz = num_sb128 * size_mul[0];
 2843|   355k|        if (cbi_sz != f->frame_thread.cbi_sz) {
  ------------------
  |  Branch (2843:13): [True: 24.3k, False: 330k]
  ------------------
 2844|  24.3k|            dav1d_free_aligned(f->frame_thread.cbi);
  ------------------
  |  |  136|  24.3k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2845|  24.3k|            f->frame_thread.cbi =
 2846|  24.3k|                dav1d_alloc_aligned(ALLOC_BLOCK, sizeof(*f->frame_thread.cbi) *
  ------------------
  |  |  134|  24.3k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2847|  24.3k|                                    cbi_sz * 32 * 32 / 4, 64);
 2848|  24.3k|            if (!f->frame_thread.cbi) {
  ------------------
  |  Branch (2848:17): [True: 0, False: 24.3k]
  ------------------
 2849|      0|                f->frame_thread.cbi_sz = 0;
 2850|      0|                goto error;
 2851|      0|            }
 2852|  24.3k|            f->frame_thread.cbi_sz = cbi_sz;
 2853|  24.3k|        }
 2854|       |
 2855|   355k|        const int cf_sz = (num_sb128 * size_mul[0]) << hbd;
 2856|   355k|        if (cf_sz != f->frame_thread.cf_sz) {
  ------------------
  |  Branch (2856:13): [True: 25.6k, False: 329k]
  ------------------
 2857|  25.6k|            dav1d_free_aligned(f->frame_thread.cf);
  ------------------
  |  |  136|  25.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2858|  25.6k|            f->frame_thread.cf =
 2859|  25.6k|                dav1d_alloc_aligned(ALLOC_COEF, (size_t)cf_sz * 128 * 128 / 2, 64);
  ------------------
  |  |  134|  25.6k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2860|  25.6k|            if (!f->frame_thread.cf) {
  ------------------
  |  Branch (2860:17): [True: 0, False: 25.6k]
  ------------------
 2861|      0|                f->frame_thread.cf_sz = 0;
 2862|      0|                goto error;
 2863|      0|            }
 2864|  25.6k|            memset(f->frame_thread.cf, 0, (size_t)cf_sz * 128 * 128 / 2);
 2865|  25.6k|            f->frame_thread.cf_sz = cf_sz;
 2866|  25.6k|        }
 2867|       |
 2868|   355k|        if (f->frame_hdr->allow_screen_content_tools) {
  ------------------
  |  Branch (2868:13): [True: 282k, False: 72.3k]
  ------------------
 2869|   282k|            const int pal_sz = num_sb128 << hbd;
 2870|   282k|            if (pal_sz != f->frame_thread.pal_sz) {
  ------------------
  |  Branch (2870:17): [True: 13.8k, False: 269k]
  ------------------
 2871|  13.8k|                dav1d_free_aligned(f->frame_thread.pal);
  ------------------
  |  |  136|  13.8k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2872|  13.8k|                f->frame_thread.pal =
 2873|  13.8k|                    dav1d_alloc_aligned(ALLOC_PAL, sizeof(*f->frame_thread.pal) *
  ------------------
  |  |  134|  13.8k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2874|  13.8k|                                        pal_sz * 16 * 16, 64);
 2875|  13.8k|                if (!f->frame_thread.pal) {
  ------------------
  |  Branch (2875:21): [True: 0, False: 13.8k]
  ------------------
 2876|      0|                    f->frame_thread.pal_sz = 0;
 2877|      0|                    goto error;
 2878|      0|                }
 2879|  13.8k|                f->frame_thread.pal_sz = pal_sz;
 2880|  13.8k|            }
 2881|       |
 2882|   282k|            const int pal_idx_sz = num_sb128 * size_mul[1];
 2883|   282k|            if (pal_idx_sz != f->frame_thread.pal_idx_sz) {
  ------------------
  |  Branch (2883:17): [True: 12.9k, False: 269k]
  ------------------
 2884|  12.9k|                dav1d_free_aligned(f->frame_thread.pal_idx);
  ------------------
  |  |  136|  12.9k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2885|  12.9k|                f->frame_thread.pal_idx =
 2886|  12.9k|                    dav1d_alloc_aligned(ALLOC_PAL, sizeof(*f->frame_thread.pal_idx) *
  ------------------
  |  |  134|  12.9k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2887|  12.9k|                                        pal_idx_sz * 128 * 128 / 8, 64);
 2888|  12.9k|                if (!f->frame_thread.pal_idx) {
  ------------------
  |  Branch (2888:21): [True: 0, False: 12.9k]
  ------------------
 2889|      0|                    f->frame_thread.pal_idx_sz = 0;
 2890|      0|                    goto error;
 2891|      0|                }
 2892|  12.9k|                f->frame_thread.pal_idx_sz = pal_idx_sz;
 2893|  12.9k|            }
 2894|   282k|        } else if (f->frame_thread.pal) {
  ------------------
  |  Branch (2894:20): [True: 1.94k, False: 70.3k]
  ------------------
 2895|  1.94k|            dav1d_freep_aligned(&f->frame_thread.pal);
 2896|  1.94k|            dav1d_freep_aligned(&f->frame_thread.pal_idx);
 2897|  1.94k|            f->frame_thread.pal_sz = f->frame_thread.pal_idx_sz = 0;
 2898|  1.94k|        }
 2899|   355k|    }
 2900|       |
 2901|       |    // update allocation of block contexts for above
 2902|   355k|    ptrdiff_t y_stride = f->cur.stride[0], uv_stride = f->cur.stride[1];
 2903|   355k|    const int has_resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
 2904|   355k|    const int need_cdef_lpf_copy = c->n_tc > 1 && has_resize;
  ------------------
  |  Branch (2904:36): [True: 355k, False: 147]
  |  Branch (2904:51): [True: 38.8k, False: 316k]
  ------------------
 2905|   355k|    if (y_stride * f->sbh * 4 != f->lf.cdef_buf_plane_sz[0] ||
  ------------------
  |  Branch (2905:9): [True: 24.3k, False: 330k]
  ------------------
 2906|   330k|        uv_stride * f->sbh * 8 != f->lf.cdef_buf_plane_sz[1] ||
  ------------------
  |  Branch (2906:9): [True: 346, False: 330k]
  ------------------
 2907|   330k|        need_cdef_lpf_copy != f->lf.need_cdef_lpf_copy ||
  ------------------
  |  Branch (2907:9): [True: 362, False: 330k]
  ------------------
 2908|   330k|        f->sbh != f->lf.cdef_buf_sbh)
  ------------------
  |  Branch (2908:9): [True: 1.17k, False: 329k]
  ------------------
 2909|  26.0k|    {
 2910|  26.0k|        dav1d_free_aligned(f->lf.cdef_line_buf);
  ------------------
  |  |  136|  26.0k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2911|  26.0k|        size_t alloc_sz = 64;
 2912|  26.0k|        alloc_sz += (size_t)llabs(y_stride) * 4 * f->sbh << need_cdef_lpf_copy;
 2913|  26.0k|        alloc_sz += (size_t)llabs(uv_stride) * 8 * f->sbh << need_cdef_lpf_copy;
 2914|  26.0k|        uint8_t *ptr = f->lf.cdef_line_buf = dav1d_alloc_aligned(ALLOC_CDEF, alloc_sz, 32);
  ------------------
  |  |  134|  26.0k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2915|  26.0k|        if (!ptr) {
  ------------------
  |  Branch (2915:13): [True: 0, False: 26.0k]
  ------------------
 2916|      0|            f->lf.cdef_buf_plane_sz[0] = f->lf.cdef_buf_plane_sz[1] = 0;
 2917|      0|            goto error;
 2918|      0|        }
 2919|       |
 2920|  26.0k|        ptr += 32;
 2921|  26.0k|        if (y_stride < 0) {
  ------------------
  |  Branch (2921:13): [True: 0, False: 26.0k]
  ------------------
 2922|      0|            f->lf.cdef_line[0][0] = ptr - y_stride * (f->sbh * 4 - 1);
 2923|      0|            f->lf.cdef_line[1][0] = ptr - y_stride * (f->sbh * 4 - 3);
 2924|  26.0k|        } else {
 2925|  26.0k|            f->lf.cdef_line[0][0] = ptr + y_stride * 0;
 2926|  26.0k|            f->lf.cdef_line[1][0] = ptr + y_stride * 2;
 2927|  26.0k|        }
 2928|  26.0k|        ptr += llabs(y_stride) * f->sbh * 4;
 2929|  26.0k|        if (uv_stride < 0) {
  ------------------
  |  Branch (2929:13): [True: 0, False: 26.0k]
  ------------------
 2930|      0|            f->lf.cdef_line[0][1] = ptr - uv_stride * (f->sbh * 8 - 1);
 2931|      0|            f->lf.cdef_line[0][2] = ptr - uv_stride * (f->sbh * 8 - 3);
 2932|      0|            f->lf.cdef_line[1][1] = ptr - uv_stride * (f->sbh * 8 - 5);
 2933|      0|            f->lf.cdef_line[1][2] = ptr - uv_stride * (f->sbh * 8 - 7);
 2934|  26.0k|        } else {
 2935|  26.0k|            f->lf.cdef_line[0][1] = ptr + uv_stride * 0;
 2936|  26.0k|            f->lf.cdef_line[0][2] = ptr + uv_stride * 2;
 2937|  26.0k|            f->lf.cdef_line[1][1] = ptr + uv_stride * 4;
 2938|  26.0k|            f->lf.cdef_line[1][2] = ptr + uv_stride * 6;
 2939|  26.0k|        }
 2940|       |
 2941|  26.0k|        if (need_cdef_lpf_copy) {
  ------------------
  |  Branch (2941:13): [True: 6.48k, False: 19.5k]
  ------------------
 2942|  6.48k|            ptr += llabs(uv_stride) * f->sbh * 8;
 2943|  6.48k|            if (y_stride < 0)
  ------------------
  |  Branch (2943:17): [True: 0, False: 6.48k]
  ------------------
 2944|      0|                f->lf.cdef_lpf_line[0] = ptr - y_stride * (f->sbh * 4 - 1);
 2945|  6.48k|            else
 2946|  6.48k|                f->lf.cdef_lpf_line[0] = ptr;
 2947|  6.48k|            ptr += llabs(y_stride) * f->sbh * 4;
 2948|  6.48k|            if (uv_stride < 0) {
  ------------------
  |  Branch (2948:17): [True: 0, False: 6.48k]
  ------------------
 2949|      0|                f->lf.cdef_lpf_line[1] = ptr - uv_stride * (f->sbh * 4 - 1);
 2950|      0|                f->lf.cdef_lpf_line[2] = ptr - uv_stride * (f->sbh * 8 - 1);
 2951|  6.48k|            } else {
 2952|  6.48k|                f->lf.cdef_lpf_line[1] = ptr;
 2953|  6.48k|                f->lf.cdef_lpf_line[2] = ptr + uv_stride * f->sbh * 4;
 2954|  6.48k|            }
 2955|  6.48k|        }
 2956|       |
 2957|  26.0k|        f->lf.cdef_buf_plane_sz[0] = (int) y_stride * f->sbh * 4;
 2958|  26.0k|        f->lf.cdef_buf_plane_sz[1] = (int) uv_stride * f->sbh * 8;
 2959|  26.0k|        f->lf.need_cdef_lpf_copy = need_cdef_lpf_copy;
 2960|  26.0k|        f->lf.cdef_buf_sbh = f->sbh;
 2961|  26.0k|    }
 2962|       |
 2963|   355k|    const int sb128 = f->seq_hdr->sb128;
 2964|   355k|    const int num_lines = c->n_tc > 1 ? f->sbh * 4 << sb128 : 12;
  ------------------
  |  Branch (2964:27): [True: 355k, False: 212]
  ------------------
 2965|   355k|    y_stride = f->sr_cur.p.stride[0], uv_stride = f->sr_cur.p.stride[1];
 2966|   355k|    if (y_stride * num_lines != f->lf.lr_buf_plane_sz[0] ||
  ------------------
  |  Branch (2966:9): [True: 24.3k, False: 330k]
  ------------------
 2967|   330k|        uv_stride * num_lines * 2 != f->lf.lr_buf_plane_sz[1])
  ------------------
  |  Branch (2967:9): [True: 346, False: 330k]
  ------------------
 2968|  24.4k|    {
 2969|  24.4k|        dav1d_free_aligned(f->lf.lr_line_buf);
  ------------------
  |  |  136|  24.4k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 2970|       |        // lr simd may overread the input, so slightly over-allocate the lpf buffer
 2971|  24.4k|        size_t alloc_sz = 128;
 2972|  24.4k|        alloc_sz += (size_t)llabs(y_stride) * num_lines;
 2973|  24.4k|        alloc_sz += (size_t)llabs(uv_stride) * num_lines * 2;
 2974|  24.4k|        uint8_t *ptr = f->lf.lr_line_buf = dav1d_alloc_aligned(ALLOC_LR, alloc_sz, 64);
  ------------------
  |  |  134|  24.4k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 2975|  24.4k|        if (!ptr) {
  ------------------
  |  Branch (2975:13): [True: 0, False: 24.4k]
  ------------------
 2976|      0|            f->lf.lr_buf_plane_sz[0] = f->lf.lr_buf_plane_sz[1] = 0;
 2977|      0|            goto error;
 2978|      0|        }
 2979|       |
 2980|  24.4k|        ptr += 64;
 2981|  24.4k|        if (y_stride < 0)
  ------------------
  |  Branch (2981:13): [True: 0, False: 24.4k]
  ------------------
 2982|      0|            f->lf.lr_lpf_line[0] = ptr - y_stride * (num_lines - 1);
 2983|  24.4k|        else
 2984|  24.4k|            f->lf.lr_lpf_line[0] = ptr;
 2985|  24.4k|        ptr += llabs(y_stride) * num_lines;
 2986|  24.4k|        if (uv_stride < 0) {
  ------------------
  |  Branch (2986:13): [True: 0, False: 24.4k]
  ------------------
 2987|      0|            f->lf.lr_lpf_line[1] = ptr - uv_stride * (num_lines * 1 - 1);
 2988|      0|            f->lf.lr_lpf_line[2] = ptr - uv_stride * (num_lines * 2 - 1);
 2989|  24.4k|        } else {
 2990|  24.4k|            f->lf.lr_lpf_line[1] = ptr;
 2991|  24.4k|            f->lf.lr_lpf_line[2] = ptr + uv_stride * num_lines;
 2992|  24.4k|        }
 2993|       |
 2994|  24.4k|        f->lf.lr_buf_plane_sz[0] = (int) y_stride * num_lines;
 2995|  24.4k|        f->lf.lr_buf_plane_sz[1] = (int) uv_stride * num_lines * 2;
 2996|  24.4k|    }
 2997|       |
 2998|       |    // update allocation for loopfilter masks
 2999|   355k|    if (num_sb128 != f->lf.mask_sz) {
  ------------------
  |  Branch (2999:9): [True: 24.0k, False: 331k]
  ------------------
 3000|  24.0k|        dav1d_free(f->lf.mask);
  ------------------
  |  |  135|  24.0k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3001|  24.0k|        dav1d_free(f->lf.level);
  ------------------
  |  |  135|  24.0k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3002|  24.0k|        f->lf.mask = dav1d_malloc(ALLOC_LF, sizeof(*f->lf.mask) * num_sb128);
  ------------------
  |  |  132|  24.0k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3003|       |        // over-allocate by 3 bytes since some of the SIMD implementations
 3004|       |        // index this from the level type and can thus over-read by up to 3
 3005|  24.0k|        f->lf.level = dav1d_malloc(ALLOC_LF, sizeof(*f->lf.level) * num_sb128 * 32 * 32 + 3);
  ------------------
  |  |  132|  24.0k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3006|  24.0k|        if (!f->lf.mask || !f->lf.level) {
  ------------------
  |  Branch (3006:13): [True: 18.4E, False: 24.0k]
  |  Branch (3006:28): [True: 0, False: 24.0k]
  ------------------
 3007|      0|            f->lf.mask_sz = 0;
 3008|      0|            goto error;
 3009|      0|        }
 3010|  24.0k|        if (c->n_fc > 1) {
  ------------------
  |  Branch (3010:13): [True: 24.0k, False: 18.4E]
  ------------------
 3011|  24.0k|            dav1d_free(f->frame_thread.b);
  ------------------
  |  |  135|  24.0k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3012|  24.0k|            f->frame_thread.b = dav1d_malloc(ALLOC_BLOCK, sizeof(*f->frame_thread.b) *
  ------------------
  |  |  132|  24.0k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3013|  24.0k|                                             num_sb128 * 32 * 32);
 3014|  24.0k|            if (!f->frame_thread.b) {
  ------------------
  |  Branch (3014:17): [True: 0, False: 24.0k]
  ------------------
 3015|      0|                f->lf.mask_sz = 0;
 3016|      0|                goto error;
 3017|      0|            }
 3018|  24.0k|        }
 3019|  24.0k|        f->lf.mask_sz = num_sb128;
 3020|  24.0k|    }
 3021|       |
 3022|   355k|    f->sr_sb128w = (f->sr_cur.p.p.w + 127) >> 7;
 3023|   355k|    const int lr_mask_sz = f->sr_sb128w * f->sb128h;
 3024|   355k|    if (lr_mask_sz != f->lf.lr_mask_sz) {
  ------------------
  |  Branch (3024:9): [True: 22.8k, False: 332k]
  ------------------
 3025|  22.8k|        dav1d_free(f->lf.lr_mask);
  ------------------
  |  |  135|  22.8k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3026|  22.8k|        f->lf.lr_mask = dav1d_malloc(ALLOC_LR, sizeof(*f->lf.lr_mask) * lr_mask_sz);
  ------------------
  |  |  132|  22.8k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3027|  22.8k|        if (!f->lf.lr_mask) {
  ------------------
  |  Branch (3027:13): [True: 0, False: 22.8k]
  ------------------
 3028|      0|            f->lf.lr_mask_sz = 0;
 3029|      0|            goto error;
 3030|      0|        }
 3031|  22.8k|        f->lf.lr_mask_sz = lr_mask_sz;
 3032|  22.8k|    }
 3033|   355k|    f->lf.restore_planes =
 3034|   355k|        ((f->frame_hdr->restoration.type[0] != DAV1D_RESTORATION_NONE) << 0) +
 3035|   355k|        ((f->frame_hdr->restoration.type[1] != DAV1D_RESTORATION_NONE) << 1) +
 3036|   355k|        ((f->frame_hdr->restoration.type[2] != DAV1D_RESTORATION_NONE) << 2);
 3037|   355k|    if (f->frame_hdr->loopfilter.sharpness != f->lf.last_sharpness) {
  ------------------
  |  Branch (3037:9): [True: 34.1k, False: 321k]
  ------------------
 3038|  34.1k|        dav1d_calc_eih(&f->lf.lim_lut, f->frame_hdr->loopfilter.sharpness);
 3039|  34.1k|        f->lf.last_sharpness = f->frame_hdr->loopfilter.sharpness;
 3040|  34.1k|    }
 3041|   355k|    dav1d_calc_lf_values(f->lf.lvl, f->frame_hdr, (int8_t[4]) { 0, 0, 0, 0 });
 3042|   355k|    memset(f->lf.mask, 0, sizeof(*f->lf.mask) * num_sb128);
 3043|       |
 3044|   355k|    const int ipred_edge_sz = f->sbh * f->sb128w << hbd;
 3045|   355k|    if (ipred_edge_sz != f->ipred_edge_sz) {
  ------------------
  |  Branch (3045:9): [True: 24.1k, False: 331k]
  ------------------
 3046|  24.1k|        dav1d_free_aligned(f->ipred_edge[0]);
  ------------------
  |  |  136|  24.1k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
 3047|  24.1k|        uint8_t *ptr = f->ipred_edge[0] =
 3048|  24.1k|            dav1d_alloc_aligned(ALLOC_IPRED, ipred_edge_sz * 128 * 3, 64);
  ------------------
  |  |  134|  24.1k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
 3049|  24.1k|        if (!ptr) {
  ------------------
  |  Branch (3049:13): [True: 0, False: 24.1k]
  ------------------
 3050|      0|            f->ipred_edge_sz = 0;
 3051|      0|            goto error;
 3052|      0|        }
 3053|  24.1k|        f->ipred_edge[1] = ptr + ipred_edge_sz * 128 * 1;
 3054|  24.1k|        f->ipred_edge[2] = ptr + ipred_edge_sz * 128 * 2;
 3055|  24.1k|        f->ipred_edge_sz = ipred_edge_sz;
 3056|  24.1k|    }
 3057|       |
 3058|   355k|    const int re_sz = f->sb128h * f->frame_hdr->tiling.cols;
 3059|   355k|    if (re_sz != f->lf.re_sz) {
  ------------------
  |  Branch (3059:9): [True: 28.2k, False: 327k]
  ------------------
 3060|  28.2k|        dav1d_free(f->lf.tx_lpf_right_edge[0]);
  ------------------
  |  |  135|  28.2k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3061|  28.2k|        f->lf.tx_lpf_right_edge[0] = dav1d_malloc(ALLOC_LF, re_sz * 32 * 2);
  ------------------
  |  |  132|  28.2k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3062|  28.2k|        if (!f->lf.tx_lpf_right_edge[0]) {
  ------------------
  |  Branch (3062:13): [True: 0, False: 28.2k]
  ------------------
 3063|      0|            f->lf.re_sz = 0;
 3064|      0|            goto error;
 3065|      0|        }
 3066|  28.2k|        f->lf.tx_lpf_right_edge[1] = f->lf.tx_lpf_right_edge[0] + re_sz * 32;
 3067|  28.2k|        f->lf.re_sz = re_sz;
 3068|  28.2k|    }
 3069|       |
 3070|       |    // init ref mvs
 3071|   355k|    if (IS_INTER_OR_SWITCH(f->frame_hdr) || f->frame_hdr->allow_intrabc) {
  ------------------
  |  |   36|   710k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 98.2k, False: 257k]
  |  |  ------------------
  ------------------
  |  Branch (3071:45): [True: 227k, False: 29.2k]
  ------------------
 3072|   325k|        const int ret =
 3073|   325k|            dav1d_refmvs_init_frame(&f->rf, f->seq_hdr, f->frame_hdr,
 3074|   325k|                                    f->refpoc, f->mvs, f->refrefpoc, f->ref_mvs,
 3075|   325k|                                    f->c->n_tc, f->c->n_fc);
 3076|   325k|        if (ret < 0) goto error;
  ------------------
  |  Branch (3076:13): [True: 0, False: 325k]
  ------------------
 3077|   325k|    }
 3078|       |
 3079|       |    // setup dequant tables
 3080|   355k|    init_quant_tables(f->seq_hdr, f->frame_hdr, f->frame_hdr->quant.yac, f->dq);
 3081|   355k|    if (f->frame_hdr->quant.qm)
  ------------------
  |  Branch (3081:9): [True: 25.8k, False: 329k]
  ------------------
 3082|   517k|        for (int i = 0; i < N_RECT_TX_SIZES; i++) {
  ------------------
  |  Branch (3082:25): [True: 491k, False: 25.8k]
  ------------------
 3083|   491k|            f->qm[i][0] = dav1d_qm_tbl[f->frame_hdr->quant.qm_y][0][i];
 3084|   491k|            f->qm[i][1] = dav1d_qm_tbl[f->frame_hdr->quant.qm_u][1][i];
 3085|   491k|            f->qm[i][2] = dav1d_qm_tbl[f->frame_hdr->quant.qm_v][1][i];
 3086|   491k|        }
 3087|   329k|    else
 3088|   329k|        memset(f->qm, 0, sizeof(f->qm));
 3089|       |
 3090|       |    // setup jnt_comp weights
 3091|   355k|    if (f->frame_hdr->switchable_comp_refs) {
  ------------------
  |  Branch (3091:9): [True: 52.7k, False: 302k]
  ------------------
 3092|   421k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3092:25): [True: 368k, False: 52.7k]
  ------------------
 3093|   368k|            const unsigned ref0poc = f->refp[i].p.frame_hdr->frame_offset;
 3094|       |
 3095|  1.46M|            for (int j = i + 1; j < 7; j++) {
  ------------------
  |  Branch (3095:33): [True: 1.09M, False: 368k]
  ------------------
 3096|  1.09M|                const unsigned ref1poc = f->refp[j].p.frame_hdr->frame_offset;
 3097|       |
 3098|  1.09M|                const unsigned d1 =
 3099|  1.09M|                    imin(abs(get_poc_diff(f->seq_hdr->order_hint_n_bits, ref0poc,
 3100|  1.09M|                                          f->cur.frame_hdr->frame_offset)), 31);
 3101|  1.09M|                const unsigned d0 =
 3102|  1.09M|                    imin(abs(get_poc_diff(f->seq_hdr->order_hint_n_bits, ref1poc,
 3103|  1.09M|                                          f->cur.frame_hdr->frame_offset)), 31);
 3104|  1.09M|                const int order = d0 <= d1;
 3105|       |
 3106|  1.09M|                static const uint8_t quant_dist_weight[3][2] = {
 3107|  1.09M|                    { 2, 3 }, { 2, 5 }, { 2, 7 }
 3108|  1.09M|                };
 3109|  1.09M|                static const uint8_t quant_dist_lookup_table[4][2] = {
 3110|  1.09M|                    { 9, 7 }, { 11, 5 }, { 12, 4 }, { 13, 3 }
 3111|  1.09M|                };
 3112|       |
 3113|  1.09M|                int k;
 3114|  2.56M|                for (k = 0; k < 3; k++) {
  ------------------
  |  Branch (3114:29): [True: 2.18M, False: 377k]
  ------------------
 3115|  2.18M|                    const int c0 = quant_dist_weight[k][order];
 3116|  2.18M|                    const int c1 = quant_dist_weight[k][!order];
 3117|  2.18M|                    const int d0_c0 = d0 * c0;
 3118|  2.18M|                    const int d1_c1 = d1 * c1;
 3119|  2.18M|                    if ((d0 > d1 && d0_c0 < d1_c1) || (d0 <= d1 && d0_c0 > d1_c1)) break;
  ------------------
  |  Branch (3119:26): [True: 760k, False: 1.42M]
  |  Branch (3119:37): [True: 192k, False: 567k]
  |  Branch (3119:56): [True: 1.43M, False: 559k]
  |  Branch (3119:68): [True: 529k, False: 904k]
  ------------------
 3120|  2.18M|                }
 3121|       |
 3122|  1.09M|                f->jnt_weights[i][j] = quant_dist_lookup_table[k][order];
 3123|  1.09M|            }
 3124|   368k|        }
 3125|  52.7k|    }
 3126|       |
 3127|       |    /* Init loopfilter pointers. Increasing NULL pointers is technically UB,
 3128|       |     * so just point the chroma pointers in 4:0:0 to the luma plane here to
 3129|       |     * avoid having additional in-loop branches in various places. We never
 3130|       |     * dereference those pointers so it doesn't really matter what they
 3131|       |     * point at, as long as the pointers are valid. */
 3132|   355k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400;
 3133|   355k|    f->lf.p[0] = f->cur.data[0];
 3134|   355k|    f->lf.p[1] = f->cur.data[has_chroma ? 1 : 0];
  ------------------
  |  Branch (3134:30): [True: 285k, False: 70.1k]
  ------------------
 3135|   355k|    f->lf.p[2] = f->cur.data[has_chroma ? 2 : 0];
  ------------------
  |  Branch (3135:30): [True: 285k, False: 70.1k]
  ------------------
 3136|   355k|    f->lf.sr_p[0] = f->sr_cur.p.data[0];
 3137|   355k|    f->lf.sr_p[1] = f->sr_cur.p.data[has_chroma ? 1 : 0];
  ------------------
  |  Branch (3137:38): [True: 285k, False: 70.1k]
  ------------------
 3138|   355k|    f->lf.sr_p[2] = f->sr_cur.p.data[has_chroma ? 2 : 0];
  ------------------
  |  Branch (3138:38): [True: 285k, False: 70.1k]
  ------------------
 3139|       |
 3140|   355k|    retval = 0;
 3141|   355k|error:
 3142|   354k|    return retval;
 3143|   355k|}
dav1d_decode_frame_init_cdf:
 3145|   302k|int dav1d_decode_frame_init_cdf(Dav1dFrameContext *const f) {
 3146|   302k|    const Dav1dContext *const c = f->c;
 3147|   302k|    int retval = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|   302k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3148|       |
 3149|   302k|    if (f->frame_hdr->refresh_context)
  ------------------
  |  Branch (3149:9): [True: 21.0k, False: 281k]
  ------------------
 3150|  21.0k|        dav1d_cdf_thread_copy(f->out_cdf.data.cdf, &f->in_cdf);
 3151|       |
 3152|       |    // parse individual tiles per tile group
 3153|   302k|    int tile_row = 0, tile_col = 0;
 3154|   302k|    f->task_thread.update_set = 0;
 3155|   598k|    for (int i = 0; i < f->n_tile_data; i++) {
  ------------------
  |  Branch (3155:21): [True: 302k, False: 295k]
  ------------------
 3156|   302k|        const uint8_t *data = f->tile[i].data.data;
 3157|   302k|        size_t size = f->tile[i].data.sz;
 3158|       |
 3159|   618k|        for (int j = f->tile[i].start; j <= f->tile[i].end; j++) {
  ------------------
  |  Branch (3159:40): [True: 323k, False: 295k]
  ------------------
 3160|   323k|            size_t tile_sz;
 3161|   323k|            if (j == f->tile[i].end) {
  ------------------
  |  Branch (3161:17): [True: 295k, False: 27.7k]
  ------------------
 3162|   295k|                tile_sz = size;
 3163|   295k|            } else {
 3164|  27.7k|                if (f->frame_hdr->tiling.n_bytes > size) goto error;
  ------------------
  |  Branch (3164:21): [True: 6.06k, False: 21.7k]
  ------------------
 3165|  21.7k|                tile_sz = 0;
 3166|  50.3k|                for (unsigned k = 0; k < f->frame_hdr->tiling.n_bytes; k++)
  ------------------
  |  Branch (3166:38): [True: 28.6k, False: 21.7k]
  ------------------
 3167|  28.6k|                    tile_sz |= (unsigned)*data++ << (k * 8);
 3168|  21.7k|                tile_sz++;
 3169|  21.7k|                size -= f->frame_hdr->tiling.n_bytes;
 3170|  21.7k|                if (tile_sz > size) goto error;
  ------------------
  |  Branch (3170:21): [True: 1.43k, False: 20.2k]
  ------------------
 3171|  21.7k|            }
 3172|       |
 3173|   315k|            setup_tile(&f->ts[j], f, data, tile_sz, tile_row, tile_col++,
 3174|   315k|                       c->n_fc > 1 ? f->frame_thread.tile_start_off[j] : 0);
  ------------------
  |  Branch (3174:24): [True: 315k, False: 20]
  ------------------
 3175|       |
 3176|   315k|            if (tile_col == f->frame_hdr->tiling.cols) {
  ------------------
  |  Branch (3176:17): [True: 306k, False: 9.06k]
  ------------------
 3177|   306k|                tile_col = 0;
 3178|   306k|                tile_row++;
 3179|   306k|            }
 3180|   315k|            if (j == f->frame_hdr->tiling.update && f->frame_hdr->refresh_context)
  ------------------
  |  Branch (3180:17): [True: 294k, False: 20.8k]
  |  Branch (3180:53): [True: 20.7k, False: 274k]
  ------------------
 3181|  20.7k|                f->task_thread.update_set = 1;
 3182|   315k|            data += tile_sz;
 3183|   315k|            size -= tile_sz;
 3184|   315k|        }
 3185|   302k|    }
 3186|       |
 3187|   295k|    if (c->n_tc > 1) {
  ------------------
  |  Branch (3187:9): [True: 294k, False: 475]
  ------------------
 3188|   294k|        const int uses_2pass = c->n_fc > 1;
 3189|  3.34M|        for (int n = 0; n < f->sb128w * f->frame_hdr->tiling.rows * (1 + uses_2pass); n++)
  ------------------
  |  Branch (3189:25): [True: 3.05M, False: 294k]
  ------------------
 3190|  3.05M|            reset_context(&f->a[n], IS_KEY_OR_INTRA(f->frame_hdr),
  ------------------
  |  |   43|  3.05M|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  3.05M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  ------------------
 3191|  18.4E|                          uses_2pass ? 1 + (n >= f->sb128w * f->frame_hdr->tiling.rows) : 0);
  ------------------
  |  Branch (3191:27): [True: 3.05M, False: 18.4E]
  ------------------
 3192|   294k|    }
 3193|       |
 3194|   295k|    retval = 0;
 3195|   301k|error:
 3196|   301k|    return retval;
 3197|   295k|}
dav1d_decode_frame_exit:
 3245|   391k|void dav1d_decode_frame_exit(Dav1dFrameContext *const f, int retval) {
 3246|   391k|    const Dav1dContext *const c = f->c;
 3247|       |
 3248|   391k|    if (f->sr_cur.p.data[0])
  ------------------
  |  Branch (3248:9): [True: 356k, False: 35.5k]
  ------------------
 3249|   391k|        atomic_init(&f->task_thread.error, 0);
 3250|       |
 3251|   391k|    if (c->n_fc > 1 && retval && f->frame_thread.cf) {
  ------------------
  |  Branch (3251:9): [True: 391k, False: 0]
  |  Branch (3251:24): [True: 220k, False: 171k]
  |  Branch (3251:34): [True: 199k, False: 20.9k]
  ------------------
 3252|   199k|        memset(f->frame_thread.cf, 0,
 3253|   199k|               (size_t)f->frame_thread.cf_sz * 128 * 128 / 2);
 3254|   199k|    }
 3255|  3.13M|    for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3255:21): [True: 2.74M, False: 391k]
  ------------------
 3256|  2.74M|        if (f->refp[i].p.frame_hdr) {
  ------------------
  |  Branch (3256:13): [True: 688k, False: 2.05M]
  ------------------
 3257|   688k|            if (!retval && c->n_fc > 1 && c->strict_std_compliance &&
  ------------------
  |  Branch (3257:17): [True: 76.4k, False: 611k]
  |  Branch (3257:28): [True: 76.4k, False: 0]
  |  Branch (3257:43): [True: 0, False: 76.4k]
  ------------------
 3258|   688k|                atomic_load(&f->refp[i].progress[1]) == FRAME_ERROR)
  ------------------
  |  |   35|      0|#define FRAME_ERROR (UINT_MAX - 1)
  ------------------
  |  Branch (3258:17): [True: 0, False: 0]
  ------------------
 3259|      0|            {
 3260|      0|                retval = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3261|      0|                atomic_store(&f->task_thread.error, 1);
 3262|      0|                atomic_store(&f->sr_cur.progress[1], FRAME_ERROR);
 3263|      0|            }
 3264|   688k|            dav1d_thread_picture_unref(&f->refp[i]);
 3265|   688k|        }
 3266|  2.74M|        dav1d_ref_dec(&f->ref_mvs_ref[i]);
 3267|  2.74M|    }
 3268|       |
 3269|   391k|    dav1d_picture_unref_internal(&f->cur);
 3270|   391k|    dav1d_thread_picture_unref(&f->sr_cur);
 3271|   391k|    dav1d_cdf_thread_unref(&f->in_cdf);
 3272|   391k|    if (f->frame_hdr && f->frame_hdr->refresh_context) {
  ------------------
  |  Branch (3272:9): [True: 371k, False: 20.5k]
  |  Branch (3272:25): [True: 59.1k, False: 312k]
  ------------------
 3273|  59.1k|        if (f->out_cdf.progress)
  ------------------
  |  Branch (3273:13): [True: 55.3k, False: 3.80k]
  ------------------
 3274|  59.1k|            atomic_store(f->out_cdf.progress, retval == 0 ? 1 : TILE_ERROR);
  ------------------
  |  Branch (3274:13): [True: 9.81k, False: 45.5k]
  ------------------
 3275|  59.1k|        dav1d_cdf_thread_unref(&f->out_cdf);
 3276|  59.1k|    }
 3277|   391k|    dav1d_ref_dec(&f->cur_segmap_ref);
 3278|   391k|    dav1d_ref_dec(&f->prev_segmap_ref);
 3279|   391k|    dav1d_ref_dec(&f->mvs_ref);
 3280|   391k|    dav1d_ref_dec(&f->seq_hdr_ref);
 3281|   391k|    dav1d_ref_dec(&f->frame_hdr_ref);
 3282|       |
 3283|   748k|    for (int i = 0; i < f->n_tile_data; i++)
  ------------------
  |  Branch (3283:21): [True: 356k, False: 391k]
  ------------------
 3284|   356k|        dav1d_data_unref_internal(&f->tile[i].data);
 3285|   391k|    f->task_thread.retval = retval;
 3286|   391k|}
dav1d_submit_frame:
 3330|   359k|int dav1d_submit_frame(Dav1dContext *const c) {
 3331|   359k|    Dav1dFrameContext *f;
 3332|   359k|    int res = -1;
 3333|       |
 3334|       |    // wait for c->out_delayed[next] and move into c->out if visible
 3335|   359k|    Dav1dThreadPicture *out_delayed;
 3336|   359k|    if (c->n_fc > 1) {
  ------------------
  |  Branch (3336:9): [True: 359k, False: 0]
  ------------------
 3337|   359k|        pthread_mutex_lock(&c->task_thread.lock);
 3338|   359k|        const unsigned next = c->frame_thread.next++;
 3339|   359k|        if (c->frame_thread.next == c->n_fc)
  ------------------
  |  Branch (3339:13): [True: 86.9k, False: 272k]
  ------------------
 3340|  86.9k|            c->frame_thread.next = 0;
 3341|       |
 3342|   359k|        f = &c->fc[next];
 3343|   477k|        while (f->n_tile_data > 0)
  ------------------
  |  Branch (3343:16): [True: 118k, False: 359k]
  ------------------
 3344|   118k|            pthread_cond_wait(&f->task_thread.cond,
 3345|   118k|                              &c->task_thread.lock);
 3346|   359k|        out_delayed = &c->frame_thread.out_delayed[next];
 3347|   359k|        if (out_delayed->p.data[0] || atomic_load(&f->task_thread.error)) {
  ------------------
  |  Branch (3347:13): [True: 339k, False: 19.6k]
  |  Branch (3347:39): [True: 2.57k, False: 17.0k]
  ------------------
 3348|   342k|            unsigned first = atomic_load(&c->task_thread.first);
 3349|   342k|            if (first + 1U < c->n_fc)
  ------------------
  |  Branch (3349:17): [True: 257k, False: 85.0k]
  ------------------
 3350|   342k|                atomic_fetch_add(&c->task_thread.first, 1U);
 3351|  85.0k|            else
 3352|   342k|                atomic_store(&c->task_thread.first, 0);
 3353|   342k|            atomic_compare_exchange_strong(&c->task_thread.reset_task_cur,
 3354|   342k|                                           &first, UINT_MAX);
 3355|   342k|            if (c->task_thread.cur && c->task_thread.cur < c->n_fc)
  ------------------
  |  Branch (3355:17): [True: 339k, False: 3.07k]
  |  Branch (3355:39): [True: 221k, False: 117k]
  ------------------
 3356|   221k|                c->task_thread.cur--;
 3357|   342k|        }
 3358|   359k|        const int error = f->task_thread.retval;
 3359|   359k|        if (error) {
  ------------------
  |  Branch (3359:13): [True: 174k, False: 184k]
  ------------------
 3360|   174k|            f->task_thread.retval = 0;
 3361|   174k|            c->cached_error = error;
 3362|   174k|            dav1d_data_props_copy(&c->cached_error_props, &out_delayed->p.m);
 3363|   174k|            dav1d_thread_picture_unref(out_delayed);
 3364|   184k|        } else if (out_delayed->p.data[0]) {
  ------------------
  |  Branch (3364:20): [True: 165k, False: 19.6k]
  ------------------
 3365|   165k|            const unsigned progress = atomic_load_explicit(&out_delayed->progress[1],
 3366|   165k|                                                           memory_order_relaxed);
 3367|   165k|            if ((out_delayed->visible || c->output_invisible_frames) &&
  ------------------
  |  Branch (3367:18): [True: 159k, False: 5.75k]
  |  Branch (3367:42): [True: 0, False: 5.75k]
  ------------------
 3368|   159k|                progress != FRAME_ERROR)
  ------------------
  |  |   35|   159k|#define FRAME_ERROR (UINT_MAX - 1)
  ------------------
  |  Branch (3368:17): [True: 155k, False: 4.31k]
  ------------------
 3369|   155k|            {
 3370|   155k|                dav1d_thread_picture_ref(&c->out, out_delayed);
 3371|   155k|                c->event_flags |= dav1d_picture_get_event_flags(out_delayed);
 3372|   155k|            }
 3373|   165k|            dav1d_thread_picture_unref(out_delayed);
 3374|   165k|        }
 3375|   359k|    } else {
 3376|      0|        f = c->fc;
 3377|      0|    }
 3378|       |
 3379|   359k|    f->seq_hdr = c->seq_hdr;
 3380|   359k|    f->seq_hdr_ref = c->seq_hdr_ref;
 3381|   359k|    dav1d_ref_inc(f->seq_hdr_ref);
 3382|   359k|    f->frame_hdr = c->frame_hdr;
 3383|   359k|    f->frame_hdr_ref = c->frame_hdr_ref;
 3384|   359k|    c->frame_hdr = NULL;
 3385|   359k|    c->frame_hdr_ref = NULL;
 3386|   359k|    f->dsp = &c->dsp[f->seq_hdr->hbd];
 3387|       |
 3388|   359k|    const int bpc = 8 + 2 * f->seq_hdr->hbd;
 3389|       |
 3390|   359k|    if (!f->dsp->ipred.intra_pred[DC_PRED]) {
  ------------------
  |  Branch (3390:9): [True: 8.57k, False: 350k]
  ------------------
 3391|  8.57k|        Dav1dDSPContext *const dsp = &c->dsp[f->seq_hdr->hbd];
 3392|       |
 3393|  8.57k|        switch (bpc) {
 3394|      0|#define assign_bitdepth_case(bd) \
 3395|      0|            dav1d_cdef_dsp_init_##bd##bpc(&dsp->cdef); \
 3396|      0|            dav1d_intra_pred_dsp_init_##bd##bpc(&dsp->ipred); \
 3397|      0|            dav1d_itx_dsp_init_##bd##bpc(&dsp->itx, bpc); \
 3398|      0|            dav1d_loop_filter_dsp_init_##bd##bpc(&dsp->lf); \
 3399|      0|            dav1d_loop_restoration_dsp_init_##bd##bpc(&dsp->lr, bpc); \
 3400|      0|            dav1d_mc_dsp_init_##bd##bpc(&dsp->mc); \
 3401|      0|            dav1d_film_grain_dsp_init_##bd##bpc(&dsp->fg); \
 3402|      0|            break
 3403|      0|#if CONFIG_8BPC
 3404|  3.46k|        case 8:
  ------------------
  |  Branch (3404:9): [True: 3.46k, False: 5.11k]
  ------------------
 3405|  3.46k|            assign_bitdepth_case(8);
  ------------------
  |  | 3395|  3.46k|            dav1d_cdef_dsp_init_##bd##bpc(&dsp->cdef); \
  |  | 3396|  3.46k|            dav1d_intra_pred_dsp_init_##bd##bpc(&dsp->ipred); \
  |  | 3397|  3.46k|            dav1d_itx_dsp_init_##bd##bpc(&dsp->itx, bpc); \
  |  | 3398|  3.46k|            dav1d_loop_filter_dsp_init_##bd##bpc(&dsp->lf); \
  |  | 3399|  3.46k|            dav1d_loop_restoration_dsp_init_##bd##bpc(&dsp->lr, bpc); \
  |  | 3400|  3.46k|            dav1d_mc_dsp_init_##bd##bpc(&dsp->mc); \
  |  | 3401|  3.46k|            dav1d_film_grain_dsp_init_##bd##bpc(&dsp->fg); \
  |  | 3402|  3.46k|            break
  ------------------
 3406|      0|#endif
 3407|      0|#if CONFIG_16BPC
 3408|  2.11k|        case 10:
  ------------------
  |  Branch (3408:9): [True: 2.11k, False: 6.46k]
  ------------------
 3409|  5.11k|        case 12:
  ------------------
  |  Branch (3409:9): [True: 2.99k, False: 5.58k]
  ------------------
 3410|  5.11k|            assign_bitdepth_case(16);
  ------------------
  |  | 3395|  5.11k|            dav1d_cdef_dsp_init_##bd##bpc(&dsp->cdef); \
  |  | 3396|  5.11k|            dav1d_intra_pred_dsp_init_##bd##bpc(&dsp->ipred); \
  |  | 3397|  5.11k|            dav1d_itx_dsp_init_##bd##bpc(&dsp->itx, bpc); \
  |  | 3398|  5.11k|            dav1d_loop_filter_dsp_init_##bd##bpc(&dsp->lf); \
  |  | 3399|  5.11k|            dav1d_loop_restoration_dsp_init_##bd##bpc(&dsp->lr, bpc); \
  |  | 3400|  5.11k|            dav1d_mc_dsp_init_##bd##bpc(&dsp->mc); \
  |  | 3401|  5.11k|            dav1d_film_grain_dsp_init_##bd##bpc(&dsp->fg); \
  |  | 3402|  5.11k|            break
  ------------------
 3411|      0|#endif
 3412|      0|#undef assign_bitdepth_case
 3413|      0|        default:
  ------------------
  |  Branch (3413:9): [True: 0, False: 8.57k]
  ------------------
 3414|      0|            dav1d_log(c, "Compiled without support for %d-bit decoding\n",
  ------------------
  |  |   44|      0|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 0]
  |  |  ------------------
  ------------------
 3415|      0|                    8 + 2 * f->seq_hdr->hbd);
 3416|      0|            res = DAV1D_ERR(ENOPROTOOPT);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3417|      0|            goto error;
 3418|  8.57k|        }
 3419|  8.57k|    }
 3420|       |
 3421|   359k|#define assign_bitdepth_case(bd) \
 3422|   359k|        f->bd_fn.recon_b_inter = dav1d_recon_b_inter_##bd##bpc; \
 3423|   359k|        f->bd_fn.recon_b_intra = dav1d_recon_b_intra_##bd##bpc; \
 3424|   359k|        f->bd_fn.filter_sbrow = dav1d_filter_sbrow_##bd##bpc; \
 3425|   359k|        f->bd_fn.filter_sbrow_deblock_cols = dav1d_filter_sbrow_deblock_cols_##bd##bpc; \
 3426|   359k|        f->bd_fn.filter_sbrow_deblock_rows = dav1d_filter_sbrow_deblock_rows_##bd##bpc; \
 3427|   359k|        f->bd_fn.filter_sbrow_cdef = dav1d_filter_sbrow_cdef_##bd##bpc; \
 3428|   359k|        f->bd_fn.filter_sbrow_resize = dav1d_filter_sbrow_resize_##bd##bpc; \
 3429|   359k|        f->bd_fn.filter_sbrow_lr = dav1d_filter_sbrow_lr_##bd##bpc; \
 3430|   359k|        f->bd_fn.backup_ipred_edge = dav1d_backup_ipred_edge_##bd##bpc; \
 3431|   359k|        f->bd_fn.read_coef_blocks = dav1d_read_coef_blocks_##bd##bpc; \
 3432|   359k|        f->bd_fn.copy_pal_block_y = dav1d_copy_pal_block_y_##bd##bpc; \
 3433|   359k|        f->bd_fn.copy_pal_block_uv = dav1d_copy_pal_block_uv_##bd##bpc; \
 3434|   359k|        f->bd_fn.read_pal_plane = dav1d_read_pal_plane_##bd##bpc; \
 3435|   359k|        f->bd_fn.read_pal_uv = dav1d_read_pal_uv_##bd##bpc
 3436|   359k|    if (!f->seq_hdr->hbd) {
  ------------------
  |  Branch (3436:9): [True: 148k, False: 210k]
  ------------------
 3437|   148k|#if CONFIG_8BPC
 3438|   148k|        assign_bitdepth_case(8);
  ------------------
  |  | 3422|   148k|        f->bd_fn.recon_b_inter = dav1d_recon_b_inter_##bd##bpc; \
  |  | 3423|   148k|        f->bd_fn.recon_b_intra = dav1d_recon_b_intra_##bd##bpc; \
  |  | 3424|   148k|        f->bd_fn.filter_sbrow = dav1d_filter_sbrow_##bd##bpc; \
  |  | 3425|   148k|        f->bd_fn.filter_sbrow_deblock_cols = dav1d_filter_sbrow_deblock_cols_##bd##bpc; \
  |  | 3426|   148k|        f->bd_fn.filter_sbrow_deblock_rows = dav1d_filter_sbrow_deblock_rows_##bd##bpc; \
  |  | 3427|   148k|        f->bd_fn.filter_sbrow_cdef = dav1d_filter_sbrow_cdef_##bd##bpc; \
  |  | 3428|   148k|        f->bd_fn.filter_sbrow_resize = dav1d_filter_sbrow_resize_##bd##bpc; \
  |  | 3429|   148k|        f->bd_fn.filter_sbrow_lr = dav1d_filter_sbrow_lr_##bd##bpc; \
  |  | 3430|   148k|        f->bd_fn.backup_ipred_edge = dav1d_backup_ipred_edge_##bd##bpc; \
  |  | 3431|   148k|        f->bd_fn.read_coef_blocks = dav1d_read_coef_blocks_##bd##bpc; \
  |  | 3432|   148k|        f->bd_fn.copy_pal_block_y = dav1d_copy_pal_block_y_##bd##bpc; \
  |  | 3433|   148k|        f->bd_fn.copy_pal_block_uv = dav1d_copy_pal_block_uv_##bd##bpc; \
  |  | 3434|   148k|        f->bd_fn.read_pal_plane = dav1d_read_pal_plane_##bd##bpc; \
  |  | 3435|   148k|        f->bd_fn.read_pal_uv = dav1d_read_pal_uv_##bd##bpc
  ------------------
 3439|   148k|#endif
 3440|   210k|    } else {
 3441|   210k|#if CONFIG_16BPC
 3442|   210k|        assign_bitdepth_case(16);
  ------------------
  |  | 3422|   210k|        f->bd_fn.recon_b_inter = dav1d_recon_b_inter_##bd##bpc; \
  |  | 3423|   210k|        f->bd_fn.recon_b_intra = dav1d_recon_b_intra_##bd##bpc; \
  |  | 3424|   210k|        f->bd_fn.filter_sbrow = dav1d_filter_sbrow_##bd##bpc; \
  |  | 3425|   210k|        f->bd_fn.filter_sbrow_deblock_cols = dav1d_filter_sbrow_deblock_cols_##bd##bpc; \
  |  | 3426|   210k|        f->bd_fn.filter_sbrow_deblock_rows = dav1d_filter_sbrow_deblock_rows_##bd##bpc; \
  |  | 3427|   210k|        f->bd_fn.filter_sbrow_cdef = dav1d_filter_sbrow_cdef_##bd##bpc; \
  |  | 3428|   210k|        f->bd_fn.filter_sbrow_resize = dav1d_filter_sbrow_resize_##bd##bpc; \
  |  | 3429|   210k|        f->bd_fn.filter_sbrow_lr = dav1d_filter_sbrow_lr_##bd##bpc; \
  |  | 3430|   210k|        f->bd_fn.backup_ipred_edge = dav1d_backup_ipred_edge_##bd##bpc; \
  |  | 3431|   210k|        f->bd_fn.read_coef_blocks = dav1d_read_coef_blocks_##bd##bpc; \
  |  | 3432|   210k|        f->bd_fn.copy_pal_block_y = dav1d_copy_pal_block_y_##bd##bpc; \
  |  | 3433|   210k|        f->bd_fn.copy_pal_block_uv = dav1d_copy_pal_block_uv_##bd##bpc; \
  |  | 3434|   210k|        f->bd_fn.read_pal_plane = dav1d_read_pal_plane_##bd##bpc; \
  |  | 3435|   210k|        f->bd_fn.read_pal_uv = dav1d_read_pal_uv_##bd##bpc
  ------------------
 3443|   210k|#endif
 3444|   210k|    }
 3445|   359k|#undef assign_bitdepth_case
 3446|       |
 3447|   359k|    int ref_coded_width[7];
 3448|   359k|    if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|   359k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 101k, False: 258k]
  |  |  ------------------
  ------------------
 3449|   101k|        if (f->frame_hdr->primary_ref_frame != DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|   101k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (3449:13): [True: 80.1k, False: 21.1k]
  ------------------
 3450|  80.1k|            const int pri_ref = f->frame_hdr->refidx[f->frame_hdr->primary_ref_frame];
 3451|  80.1k|            if (!c->refs[pri_ref].p.p.data[0]) {
  ------------------
  |  Branch (3451:17): [True: 343, False: 79.8k]
  ------------------
 3452|    343|                res = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|    343|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3453|    343|                goto error;
 3454|    343|            }
 3455|  80.1k|        }
 3456|   790k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3456:25): [True: 692k, False: 98.3k]
  ------------------
 3457|   692k|            const int refidx = f->frame_hdr->refidx[i];
 3458|   692k|            if (!c->refs[refidx].p.p.data[0] ||
  ------------------
  |  Branch (3458:17): [True: 233, False: 691k]
  ------------------
 3459|   691k|                f->frame_hdr->width[0] * 2 < c->refs[refidx].p.p.p.w ||
  ------------------
  |  Branch (3459:17): [True: 548, False: 691k]
  ------------------
 3460|   691k|                f->frame_hdr->height * 2 < c->refs[refidx].p.p.p.h ||
  ------------------
  |  Branch (3460:17): [True: 1.25k, False: 690k]
  ------------------
 3461|   690k|                f->frame_hdr->width[0] > c->refs[refidx].p.p.p.w * 16 ||
  ------------------
  |  Branch (3461:17): [True: 347, False: 689k]
  ------------------
 3462|   689k|                f->frame_hdr->height > c->refs[refidx].p.p.p.h * 16 ||
  ------------------
  |  Branch (3462:17): [True: 284, False: 689k]
  ------------------
 3463|   689k|                f->seq_hdr->layout != c->refs[refidx].p.p.p.layout ||
  ------------------
  |  Branch (3463:17): [True: 0, False: 689k]
  ------------------
 3464|   689k|                bpc != c->refs[refidx].p.p.p.bpc)
  ------------------
  |  Branch (3464:17): [True: 0, False: 689k]
  ------------------
 3465|  2.66k|            {
 3466|  4.00k|                for (int j = 0; j < i; j++)
  ------------------
  |  Branch (3466:33): [True: 1.33k, False: 2.66k]
  ------------------
 3467|  1.33k|                    dav1d_thread_picture_unref(&f->refp[j]);
 3468|  2.66k|                res = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  2.66k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3469|  2.66k|                goto error;
 3470|  2.66k|            }
 3471|   689k|            dav1d_thread_picture_ref(&f->refp[i], &c->refs[refidx].p);
 3472|   689k|            ref_coded_width[i] = c->refs[refidx].p.p.frame_hdr->width[0];
 3473|   689k|            if (f->frame_hdr->width[0] != c->refs[refidx].p.p.p.w ||
  ------------------
  |  Branch (3473:17): [True: 209k, False: 480k]
  ------------------
 3474|   480k|                f->frame_hdr->height != c->refs[refidx].p.p.p.h)
  ------------------
  |  Branch (3474:17): [True: 2.79k, False: 477k]
  ------------------
 3475|   211k|            {
 3476|   211k|#define scale_fac(ref_sz, this_sz) \
 3477|   211k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
 3478|   211k|                f->svc[i][0].scale = scale_fac(c->refs[refidx].p.p.p.w,
  ------------------
  |  | 3477|   211k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
  ------------------
 3479|   211k|                                               f->frame_hdr->width[0]);
 3480|   211k|                f->svc[i][1].scale = scale_fac(c->refs[refidx].p.p.p.h,
  ------------------
  |  | 3477|   211k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
  ------------------
 3481|   211k|                                               f->frame_hdr->height);
 3482|   211k|                f->svc[i][0].step = (f->svc[i][0].scale + 8) >> 4;
 3483|   211k|                f->svc[i][1].step = (f->svc[i][1].scale + 8) >> 4;
 3484|   477k|            } else {
 3485|   477k|                f->svc[i][0].scale = f->svc[i][1].scale = 0;
 3486|   477k|            }
 3487|   689k|            f->gmv_warp_allowed[i] = f->frame_hdr->gmv[i].type > DAV1D_WM_TYPE_TRANSLATION &&
  ------------------
  |  Branch (3487:38): [True: 30.3k, False: 659k]
  ------------------
 3488|  30.3k|                                     !f->frame_hdr->force_integer_mv &&
  ------------------
  |  Branch (3488:38): [True: 27.4k, False: 2.95k]
  ------------------
 3489|  27.4k|                                     !dav1d_get_shear_params(&f->frame_hdr->gmv[i]) &&
  ------------------
  |  Branch (3489:38): [True: 25.8k, False: 1.59k]
  ------------------
 3490|  25.8k|                                     !f->svc[i][0].scale;
  ------------------
  |  Branch (3490:38): [True: 12.1k, False: 13.6k]
  ------------------
 3491|   689k|        }
 3492|   100k|    }
 3493|       |
 3494|       |    // setup entropy
 3495|   356k|    if (f->frame_hdr->primary_ref_frame == DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|   356k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (3495:9): [True: 278k, False: 77.3k]
  ------------------
 3496|   278k|        dav1d_cdf_thread_init_static(&f->in_cdf, f->frame_hdr->quant.yac);
 3497|   278k|    } else {
 3498|  77.3k|        const int pri_ref = f->frame_hdr->refidx[f->frame_hdr->primary_ref_frame];
 3499|  77.3k|        dav1d_cdf_thread_ref(&f->in_cdf, &c->cdf[pri_ref]);
 3500|  77.3k|    }
 3501|   356k|    if (f->frame_hdr->refresh_context) {
  ------------------
  |  Branch (3501:9): [True: 55.3k, False: 301k]
  ------------------
 3502|  55.3k|        res = dav1d_cdf_thread_alloc(c, &f->out_cdf, c->n_fc > 1);
 3503|  55.3k|        if (res < 0) goto error;
  ------------------
  |  Branch (3503:13): [True: 0, False: 55.3k]
  ------------------
 3504|  55.3k|    }
 3505|       |
 3506|       |    // FIXME qsort so tiles are in order (for frame threading)
 3507|   356k|    if (f->n_tile_data_alloc < c->n_tile_data) {
  ------------------
  |  Branch (3507:9): [True: 16.9k, False: 339k]
  ------------------
 3508|  16.9k|        dav1d_free(f->tile);
  ------------------
  |  |  135|  16.9k|#define dav1d_free(ptr) free(ptr)
  ------------------
 3509|  16.9k|        assert(c->n_tile_data < INT_MAX / (int)sizeof(*f->tile));
  ------------------
  |  Branch (3509:9): [True: 16.9k, False: 0]
  ------------------
 3510|  16.9k|        f->tile = dav1d_malloc(ALLOC_TILE, c->n_tile_data * sizeof(*f->tile));
  ------------------
  |  |  132|  16.9k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 3511|  16.9k|        if (!f->tile) {
  ------------------
  |  Branch (3511:13): [True: 0, False: 16.9k]
  ------------------
 3512|      0|            f->n_tile_data_alloc = f->n_tile_data = 0;
 3513|      0|            res = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3514|      0|            goto error;
 3515|      0|        }
 3516|  16.9k|        f->n_tile_data_alloc = c->n_tile_data;
 3517|  16.9k|    }
 3518|   356k|    memcpy(f->tile, c->tile, c->n_tile_data * sizeof(*f->tile));
 3519|   356k|    memset(c->tile, 0, c->n_tile_data * sizeof(*c->tile));
 3520|   356k|    f->n_tile_data = c->n_tile_data;
 3521|   356k|    c->n_tile_data = 0;
 3522|       |
 3523|       |    // allocate frame
 3524|   356k|    res = dav1d_thread_picture_alloc(c, f, bpc);
 3525|   356k|    if (res < 0) goto error;
  ------------------
  |  Branch (3525:9): [True: 0, False: 356k]
  ------------------
 3526|       |
 3527|   356k|    if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (3527:9): [True: 39.1k, False: 317k]
  ------------------
 3528|  39.1k|        res = dav1d_picture_alloc_copy(c, &f->cur, f->frame_hdr->width[0], &f->sr_cur.p);
 3529|  39.1k|        if (res < 0) goto error;
  ------------------
  |  Branch (3529:13): [True: 0, False: 39.1k]
  ------------------
 3530|   317k|    } else {
 3531|   317k|        dav1d_picture_ref(&f->cur, &f->sr_cur.p);
 3532|   317k|    }
 3533|       |
 3534|   356k|    if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (3534:9): [True: 39.1k, False: 317k]
  ------------------
 3535|  39.1k|        f->resize_step[0] = scale_fac(f->cur.p.w, f->sr_cur.p.p.w);
  ------------------
  |  | 3477|  39.1k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
  ------------------
 3536|  39.1k|        const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 3537|  39.1k|        const int in_cw = (f->cur.p.w + ss_hor) >> ss_hor;
 3538|  39.1k|        const int out_cw = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
 3539|  39.1k|        f->resize_step[1] = scale_fac(in_cw, out_cw);
  ------------------
  |  | 3477|  39.1k|    ((((ref_sz) << 14) + ((this_sz) >> 1)) / (this_sz))
  ------------------
 3540|  39.1k|#undef scale_fac
 3541|  39.1k|        f->resize_start[0] = get_upscale_x0(f->cur.p.w, f->sr_cur.p.p.w, f->resize_step[0]);
 3542|  39.1k|        f->resize_start[1] = get_upscale_x0(in_cw, out_cw, f->resize_step[1]);
 3543|  39.1k|    }
 3544|       |
 3545|       |    // move f->cur into output queue
 3546|   356k|    if (c->n_fc == 1) {
  ------------------
  |  Branch (3546:9): [True: 0, False: 356k]
  ------------------
 3547|      0|        if (f->frame_hdr->show_frame || c->output_invisible_frames) {
  ------------------
  |  Branch (3547:13): [True: 0, False: 0]
  |  Branch (3547:41): [True: 0, False: 0]
  ------------------
 3548|      0|            dav1d_thread_picture_ref(&c->out, &f->sr_cur);
 3549|      0|            c->event_flags |= dav1d_picture_get_event_flags(&f->sr_cur);
 3550|      0|        }
 3551|   356k|    } else {
 3552|   356k|        dav1d_thread_picture_ref(out_delayed, &f->sr_cur);
 3553|   356k|    }
 3554|       |
 3555|   356k|    f->w4 = (f->frame_hdr->width[0] + 3) >> 2;
 3556|   356k|    f->h4 = (f->frame_hdr->height + 3) >> 2;
 3557|   356k|    f->bw = ((f->frame_hdr->width[0] + 7) >> 3) << 1;
 3558|   356k|    f->bh = ((f->frame_hdr->height + 7) >> 3) << 1;
 3559|   356k|    f->sb128w = (f->bw + 31) >> 5;
 3560|   356k|    f->sb128h = (f->bh + 31) >> 5;
 3561|   356k|    f->sb_shift = 4 + f->seq_hdr->sb128;
 3562|   356k|    f->sb_step = 16 << f->seq_hdr->sb128;
 3563|   356k|    f->sbh = (f->bh + f->sb_step - 1) >> f->sb_shift;
 3564|   356k|    f->b4_stride = (f->bw + 31) & ~31;
 3565|   356k|    f->bitdepth_max = (1 << f->cur.p.bpc) - 1;
 3566|   356k|    atomic_init(&f->task_thread.error, 0);
 3567|   356k|    const int uses_2pass = c->n_fc > 1;
 3568|   356k|    const int cols = f->frame_hdr->tiling.cols;
 3569|   356k|    const int rows = f->frame_hdr->tiling.rows;
 3570|   356k|    atomic_store(&f->task_thread.task_counter,
 3571|   356k|                 (cols * rows + f->sbh) << uses_2pass);
 3572|       |
 3573|       |    // ref_mvs
 3574|   356k|    if (IS_INTER_OR_SWITCH(f->frame_hdr) || f->frame_hdr->allow_intrabc) {
  ------------------
  |  |   36|   712k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 98.3k, False: 258k]
  |  |  ------------------
  ------------------
  |  Branch (3574:45): [True: 228k, False: 29.5k]
  ------------------
 3575|   326k|        f->mvs_ref = dav1d_ref_create_using_pool(c->refmvs_pool,
 3576|   326k|            sizeof(*f->mvs) * f->sb128h * 16 * (f->b4_stride >> 1));
 3577|   326k|        if (!f->mvs_ref) {
  ------------------
  |  Branch (3577:13): [True: 0, False: 326k]
  ------------------
 3578|      0|            res = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3579|      0|            goto error;
 3580|      0|        }
 3581|   326k|        f->mvs = f->mvs_ref->data;
 3582|   326k|        if (!f->frame_hdr->allow_intrabc) {
  ------------------
  |  Branch (3582:13): [True: 98.3k, False: 228k]
  ------------------
 3583|   786k|            for (int i = 0; i < 7; i++)
  ------------------
  |  Branch (3583:29): [True: 688k, False: 98.3k]
  ------------------
 3584|   688k|                f->refpoc[i] = f->refp[i].p.frame_hdr->frame_offset;
 3585|   228k|        } else {
 3586|   228k|            memset(f->refpoc, 0, sizeof(f->refpoc));
 3587|   228k|        }
 3588|   326k|        if (f->frame_hdr->use_ref_frame_mvs) {
  ------------------
  |  Branch (3588:13): [True: 75.8k, False: 250k]
  ------------------
 3589|   606k|            for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3589:29): [True: 530k, False: 75.8k]
  ------------------
 3590|   530k|                const int refidx = f->frame_hdr->refidx[i];
 3591|   530k|                const int ref_w = ((ref_coded_width[i] + 7) >> 3) << 1;
 3592|   530k|                const int ref_h = ((f->refp[i].p.p.h + 7) >> 3) << 1;
 3593|   530k|                if (c->refs[refidx].refmvs != NULL &&
  ------------------
  |  Branch (3593:21): [True: 434k, False: 95.9k]
  ------------------
 3594|   434k|                    ref_w == f->bw && ref_h == f->bh)
  ------------------
  |  Branch (3594:21): [True: 428k, False: 6.29k]
  |  Branch (3594:39): [True: 428k, False: 516]
  ------------------
 3595|   428k|                {
 3596|   428k|                    f->ref_mvs_ref[i] = c->refs[refidx].refmvs;
 3597|   428k|                    dav1d_ref_inc(f->ref_mvs_ref[i]);
 3598|   428k|                    f->ref_mvs[i] = c->refs[refidx].refmvs->data;
 3599|   428k|                } else {
 3600|   102k|                    f->ref_mvs[i] = NULL;
 3601|   102k|                    f->ref_mvs_ref[i] = NULL;
 3602|   102k|                }
 3603|   530k|                memcpy(f->refrefpoc[i], c->refs[refidx].refpoc,
 3604|   530k|                       sizeof(*f->refrefpoc));
 3605|   530k|            }
 3606|   250k|        } else {
 3607|   250k|            memset(f->ref_mvs_ref, 0, sizeof(f->ref_mvs_ref));
 3608|   250k|        }
 3609|   326k|    } else {
 3610|  29.5k|        f->mvs_ref = NULL;
 3611|  29.5k|        memset(f->ref_mvs_ref, 0, sizeof(f->ref_mvs_ref));
 3612|  29.5k|    }
 3613|       |
 3614|       |    // segmap
 3615|   356k|    if (f->frame_hdr->segmentation.enabled) {
  ------------------
  |  Branch (3615:9): [True: 19.6k, False: 336k]
  ------------------
 3616|       |        // By default, the previous segmentation map is not initialised.
 3617|  19.6k|        f->prev_segmap_ref = NULL;
 3618|  19.6k|        f->prev_segmap = NULL;
 3619|       |
 3620|       |        // We might need a previous frame's segmentation map. This
 3621|       |        // happens if there is either no update or a temporal update.
 3622|  19.6k|        if (f->frame_hdr->segmentation.temporal || !f->frame_hdr->segmentation.update_map) {
  ------------------
  |  Branch (3622:13): [True: 8.32k, False: 11.2k]
  |  Branch (3622:52): [True: 6.79k, False: 4.50k]
  ------------------
 3623|  15.1k|            const int pri_ref = f->frame_hdr->primary_ref_frame;
 3624|  15.1k|            assert(pri_ref != DAV1D_PRIMARY_REF_NONE);
  ------------------
  |  Branch (3624:13): [True: 15.1k, False: 0]
  ------------------
 3625|  15.1k|            const int ref_w = ((ref_coded_width[pri_ref] + 7) >> 3) << 1;
 3626|  15.1k|            const int ref_h = ((f->refp[pri_ref].p.p.h + 7) >> 3) << 1;
 3627|  15.1k|            if (ref_w == f->bw && ref_h == f->bh) {
  ------------------
  |  Branch (3627:17): [True: 13.1k, False: 1.92k]
  |  Branch (3627:35): [True: 12.0k, False: 1.17k]
  ------------------
 3628|  12.0k|                f->prev_segmap_ref = c->refs[f->frame_hdr->refidx[pri_ref]].segmap;
 3629|  12.0k|                if (f->prev_segmap_ref) {
  ------------------
  |  Branch (3629:21): [True: 10.4k, False: 1.54k]
  ------------------
 3630|  10.4k|                    dav1d_ref_inc(f->prev_segmap_ref);
 3631|  10.4k|                    f->prev_segmap = f->prev_segmap_ref->data;
 3632|  10.4k|                }
 3633|  12.0k|            }
 3634|  15.1k|        }
 3635|       |
 3636|  19.6k|        if (f->frame_hdr->segmentation.update_map) {
  ------------------
  |  Branch (3636:13): [True: 12.8k, False: 6.79k]
  ------------------
 3637|       |            // We're updating an existing map, but need somewhere to
 3638|       |            // put the new values. Allocate them here (the data
 3639|       |            // actually gets set elsewhere)
 3640|  12.8k|            f->cur_segmap_ref = dav1d_ref_create_using_pool(c->segmap_pool,
 3641|  12.8k|                sizeof(*f->cur_segmap) * f->b4_stride * 32 * f->sb128h);
 3642|  12.8k|            if (!f->cur_segmap_ref) {
  ------------------
  |  Branch (3642:17): [True: 0, False: 12.8k]
  ------------------
 3643|      0|                dav1d_ref_dec(&f->prev_segmap_ref);
 3644|      0|                res = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3645|      0|                goto error;
 3646|      0|            }
 3647|  12.8k|            f->cur_segmap = f->cur_segmap_ref->data;
 3648|  12.8k|        } else if (f->prev_segmap_ref) {
  ------------------
  |  Branch (3648:20): [True: 4.62k, False: 2.17k]
  ------------------
 3649|       |            // We're not updating an existing map, and we have a valid
 3650|       |            // reference. Use that.
 3651|  4.62k|            f->cur_segmap_ref = f->prev_segmap_ref;
 3652|  4.62k|            dav1d_ref_inc(f->cur_segmap_ref);
 3653|  4.62k|            f->cur_segmap = f->prev_segmap_ref->data;
 3654|  4.62k|        } else {
 3655|       |            // We need to make a new map. Allocate one here and zero it out.
 3656|  2.17k|            const size_t segmap_size = sizeof(*f->cur_segmap) * f->b4_stride * 32 * f->sb128h;
 3657|  2.17k|            f->cur_segmap_ref = dav1d_ref_create_using_pool(c->segmap_pool, segmap_size);
 3658|  2.17k|            if (!f->cur_segmap_ref) {
  ------------------
  |  Branch (3658:17): [True: 0, False: 2.17k]
  ------------------
 3659|      0|                res = DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 3660|      0|                goto error;
 3661|      0|            }
 3662|  2.17k|            f->cur_segmap = f->cur_segmap_ref->data;
 3663|  2.17k|            memset(f->cur_segmap, 0, segmap_size);
 3664|  2.17k|        }
 3665|   336k|    } else {
 3666|   336k|        f->cur_segmap = NULL;
 3667|   336k|        f->cur_segmap_ref = NULL;
 3668|   336k|        f->prev_segmap_ref = NULL;
 3669|   336k|    }
 3670|       |
 3671|       |    // update references etc.
 3672|   356k|    const unsigned refresh_frame_flags = f->frame_hdr->refresh_frame_flags;
 3673|  3.20M|    for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (3673:21): [True: 2.85M, False: 356k]
  ------------------
 3674|  2.85M|        if (refresh_frame_flags & (1 << i)) {
  ------------------
  |  Branch (3674:13): [True: 2.40M, False: 444k]
  ------------------
 3675|  2.40M|            if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (3675:17): [True: 2.30M, False: 106k]
  ------------------
 3676|  2.30M|                dav1d_thread_picture_unref(&c->refs[i].p);
 3677|  2.40M|            dav1d_thread_picture_ref(&c->refs[i].p, &f->sr_cur);
 3678|       |
 3679|  2.40M|            dav1d_cdf_thread_unref(&c->cdf[i]);
 3680|  2.40M|            if (f->frame_hdr->refresh_context) {
  ------------------
  |  Branch (3680:17): [True: 195k, False: 2.21M]
  ------------------
 3681|   195k|                dav1d_cdf_thread_ref(&c->cdf[i], &f->out_cdf);
 3682|  2.21M|            } else {
 3683|  2.21M|                dav1d_cdf_thread_ref(&c->cdf[i], &f->in_cdf);
 3684|  2.21M|            }
 3685|       |
 3686|  2.40M|            dav1d_ref_dec(&c->refs[i].segmap);
 3687|  2.40M|            c->refs[i].segmap = f->cur_segmap_ref;
 3688|  2.40M|            if (f->cur_segmap_ref)
  ------------------
  |  Branch (3688:17): [True: 101k, False: 2.30M]
  ------------------
 3689|   101k|                dav1d_ref_inc(f->cur_segmap_ref);
 3690|  2.40M|            dav1d_ref_dec(&c->refs[i].refmvs);
 3691|  2.40M|            if (!f->frame_hdr->allow_intrabc) {
  ------------------
  |  Branch (3691:17): [True: 581k, False: 1.82M]
  ------------------
 3692|   581k|                c->refs[i].refmvs = f->mvs_ref;
 3693|   581k|                if (f->mvs_ref)
  ------------------
  |  Branch (3693:21): [True: 353k, False: 228k]
  ------------------
 3694|   353k|                    dav1d_ref_inc(f->mvs_ref);
 3695|   581k|            }
 3696|  2.40M|            memcpy(c->refs[i].refpoc, f->refpoc, sizeof(f->refpoc));
 3697|  2.40M|        }
 3698|  2.85M|    }
 3699|       |
 3700|   356k|    if (c->n_fc == 1) {
  ------------------
  |  Branch (3700:9): [True: 0, False: 356k]
  ------------------
 3701|      0|        if ((res = dav1d_decode_frame(f)) < 0) {
  ------------------
  |  Branch (3701:13): [True: 0, False: 0]
  ------------------
 3702|      0|            dav1d_thread_picture_unref(&c->out);
 3703|      0|            for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (3703:29): [True: 0, False: 0]
  ------------------
 3704|      0|                if (refresh_frame_flags & (1 << i)) {
  ------------------
  |  Branch (3704:21): [True: 0, False: 0]
  ------------------
 3705|      0|                    if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (3705:25): [True: 0, False: 0]
  ------------------
 3706|      0|                        dav1d_thread_picture_unref(&c->refs[i].p);
 3707|      0|                    dav1d_cdf_thread_unref(&c->cdf[i]);
 3708|      0|                    dav1d_ref_dec(&c->refs[i].segmap);
 3709|      0|                    dav1d_ref_dec(&c->refs[i].refmvs);
 3710|      0|                }
 3711|      0|            }
 3712|      0|            goto error;
 3713|      0|        }
 3714|   356k|    } else {
 3715|   356k|        dav1d_task_frame_init(f);
 3716|   356k|        pthread_mutex_unlock(&c->task_thread.lock);
 3717|   356k|    }
 3718|       |
 3719|   356k|    return 0;
 3720|  3.01k|error:
 3721|  3.01k|    atomic_init(&f->task_thread.error, 1);
 3722|  3.01k|    dav1d_cdf_thread_unref(&f->in_cdf);
 3723|  3.01k|    if (f->frame_hdr->refresh_context)
  ------------------
  |  Branch (3723:9): [True: 2.49k, False: 512]
  ------------------
 3724|  2.49k|        dav1d_cdf_thread_unref(&f->out_cdf);
 3725|  24.0k|    for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (3725:21): [True: 21.0k, False: 3.01k]
  ------------------
 3726|  21.0k|        if (f->refp[i].p.frame_hdr)
  ------------------
  |  Branch (3726:13): [True: 0, False: 21.0k]
  ------------------
 3727|      0|            dav1d_thread_picture_unref(&f->refp[i]);
 3728|  21.0k|        dav1d_ref_dec(&f->ref_mvs_ref[i]);
 3729|  21.0k|    }
 3730|  3.01k|    if (c->n_fc == 1)
  ------------------
  |  Branch (3730:9): [True: 0, False: 3.01k]
  ------------------
 3731|      0|        dav1d_thread_picture_unref(&c->out);
 3732|  3.01k|    else
 3733|  3.01k|        dav1d_thread_picture_unref(out_delayed);
 3734|  3.01k|    dav1d_picture_unref_internal(&f->cur);
 3735|  3.01k|    dav1d_thread_picture_unref(&f->sr_cur);
 3736|  3.01k|    dav1d_ref_dec(&f->mvs_ref);
 3737|  3.01k|    dav1d_ref_dec(&f->seq_hdr_ref);
 3738|  3.01k|    dav1d_ref_dec(&f->frame_hdr_ref);
 3739|  3.01k|    dav1d_data_props_copy(&c->cached_error_props, &c->in.m);
 3740|       |
 3741|  3.01k|    for (int i = 0; i < f->n_tile_data; i++)
  ------------------
  |  Branch (3741:21): [True: 0, False: 3.01k]
  ------------------
 3742|      0|        dav1d_data_unref_internal(&f->tile[i].data);
 3743|  3.01k|    f->n_tile_data = 0;
 3744|       |
 3745|  3.01k|    if (c->n_fc > 1)
  ------------------
  |  Branch (3745:9): [True: 3.01k, False: 0]
  ------------------
 3746|  3.01k|        pthread_mutex_unlock(&c->task_thread.lock);
 3747|       |
 3748|  3.01k|    return res;
 3749|   356k|}
decode.c:reset_context:
 2392|  7.17M|static void reset_context(BlockContext *const ctx, const int keyframe, const int pass) {
 2393|  7.17M|    memset(ctx->intra, keyframe, sizeof(ctx->intra));
 2394|  7.17M|    memset(ctx->uvmode, DC_PRED, sizeof(ctx->uvmode));
 2395|  7.17M|    if (keyframe)
  ------------------
  |  Branch (2395:9): [True: 2.58M, False: 4.59M]
  ------------------
 2396|  2.58M|        memset(ctx->mode, DC_PRED, sizeof(ctx->mode));
 2397|       |
 2398|  7.17M|    if (pass == 2) return;
  ------------------
  |  Branch (2398:9): [True: 3.53M, False: 3.64M]
  ------------------
 2399|       |
 2400|  3.64M|    memset(ctx->partition, 0, sizeof(ctx->partition));
 2401|  3.64M|    memset(ctx->skip, 0, sizeof(ctx->skip));
 2402|  3.64M|    memset(ctx->skip_mode, 0, sizeof(ctx->skip_mode));
 2403|  3.64M|    memset(ctx->tx_lpf_y, 2, sizeof(ctx->tx_lpf_y));
 2404|  3.64M|    memset(ctx->tx_lpf_uv, 1, sizeof(ctx->tx_lpf_uv));
 2405|  3.64M|    memset(ctx->tx_intra, -1, sizeof(ctx->tx_intra));
 2406|  3.64M|    memset(ctx->tx, TX_64X64, sizeof(ctx->tx));
 2407|  3.64M|    if (!keyframe) {
  ------------------
  |  Branch (2407:9): [True: 2.31M, False: 1.32M]
  ------------------
 2408|  2.31M|        memset(ctx->ref, -1, sizeof(ctx->ref));
 2409|  2.31M|        memset(ctx->comp_type, 0, sizeof(ctx->comp_type));
 2410|  2.31M|        memset(ctx->mode, NEARESTMV, sizeof(ctx->mode));
 2411|  2.31M|    }
 2412|  3.64M|    memset(ctx->lcoef, 0x40, sizeof(ctx->lcoef));
 2413|  3.64M|    memset(ctx->ccoef, 0x40, sizeof(ctx->ccoef));
 2414|  3.64M|    memset(ctx->filter, DAV1D_N_SWITCHABLE_FILTERS, sizeof(ctx->filter));
 2415|  3.64M|    memset(ctx->seg_pred, 0, sizeof(ctx->seg_pred));
 2416|  3.64M|    memset(ctx->pal_sz, 0, sizeof(ctx->pal_sz));
 2417|  3.64M|}
decode.c:decode_sb:
 2121|  13.1M|{
 2122|  13.1M|    const Dav1dFrameContext *const f = t->f;
 2123|  13.1M|    Dav1dTileState *const ts = t->ts;
 2124|  13.1M|    const int hsz = 16 >> bl;
 2125|  13.1M|    const int have_h_split = f->bw > t->bx + hsz;
 2126|  13.1M|    const int have_v_split = f->bh > t->by + hsz;
 2127|       |
 2128|  13.1M|    if (!have_h_split && !have_v_split) {
  ------------------
  |  Branch (2128:9): [True: 899k, False: 12.2M]
  |  Branch (2128:26): [True: 314k, False: 585k]
  ------------------
 2129|   314k|        assert(bl < BL_8X8);
  ------------------
  |  Branch (2129:9): [True: 314k, False: 0]
  ------------------
 2130|   314k|        return decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 0));
  ------------------
  |  |   51|   314k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
 2131|   314k|    }
 2132|       |
 2133|  12.8M|    uint16_t *pc;
 2134|  12.8M|    enum BlockPartition bp;
 2135|  12.8M|    int ctx, bx8, by8;
 2136|  12.8M|    if (t->frame_thread.pass != 2) {
  ------------------
  |  Branch (2136:9): [True: 8.89M, False: 3.91M]
  ------------------
 2137|  8.89M|        if (0 && bl == BL_64X64)
  ------------------
  |  Branch (2137:13): [Folded, False: 8.89M]
  |  Branch (2137:18): [True: 0, False: 0]
  ------------------
 2138|      0|            printf("poc=%d,y=%d,x=%d,bl=%d,r=%d\n",
 2139|      0|                   f->frame_hdr->frame_offset, t->by, t->bx, bl, ts->msac.rng);
 2140|  8.89M|        bx8 = (t->bx & 31) >> 1;
 2141|  8.89M|        by8 = (t->by & 31) >> 1;
 2142|  8.89M|        ctx = get_partition_ctx(t->a, &t->l, bl, by8, bx8);
 2143|  8.89M|        pc = ts->cdf.m.partition[bl][ctx];
 2144|  8.89M|    }
 2145|       |
 2146|  12.8M|    if (have_h_split && have_v_split) {
  ------------------
  |  Branch (2146:9): [True: 12.2M, False: 582k]
  |  Branch (2146:25): [True: 11.8M, False: 416k]
  ------------------
 2147|  11.8M|        if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (2147:13): [True: 3.66M, False: 8.13M]
  ------------------
 2148|  3.66M|            const Av1Block *const b = &f->frame_thread.b[t->by * f->b4_stride + t->bx];
 2149|  3.66M|            bp = b->bl == bl ? b->bp : PARTITION_SPLIT;
  ------------------
  |  Branch (2149:18): [True: 3.32M, False: 343k]
  ------------------
 2150|  8.13M|        } else {
 2151|  8.13M|            bp = dav1d_msac_decode_symbol_adapt16(&ts->msac, pc,
  ------------------
  |  |   57|  8.13M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 2152|  8.13M|                                                  dav1d_partition_type_count[bl]);
 2153|  8.13M|            if (f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I422 &&
  ------------------
  |  Branch (2153:17): [True: 72.4k, False: 8.06M]
  ------------------
 2154|  72.4k|                (bp == PARTITION_V || bp == PARTITION_V4 ||
  ------------------
  |  Branch (2154:18): [True: 470, False: 72.0k]
  |  Branch (2154:39): [True: 1.37k, False: 70.6k]
  ------------------
 2155|  70.6k|                 bp == PARTITION_T_LEFT_SPLIT || bp == PARTITION_T_RIGHT_SPLIT))
  ------------------
  |  Branch (2155:18): [True: 374, False: 70.2k]
  |  Branch (2155:50): [True: 55.7k, False: 14.5k]
  ------------------
 2156|  57.9k|            {
 2157|  57.9k|                return 1;
 2158|  57.9k|            }
 2159|  8.08M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  8.08M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 8.08M]
  |  |  ------------------
  |  |   35|  8.08M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  8.08M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2160|      0|                printf("poc=%d,y=%d,x=%d,bl=%d,ctx=%d,bp=%d: r=%d\n",
 2161|      0|                       f->frame_hdr->frame_offset, t->by, t->bx, bl, ctx, bp,
 2162|      0|                       ts->msac.rng);
 2163|  8.08M|        }
 2164|  11.7M|        const uint8_t *const b = dav1d_block_sizes[bl][bp];
 2165|       |
 2166|  11.7M|        switch (bp) {
 2167|  7.01M|        case PARTITION_NONE:
  ------------------
  |  Branch (2167:9): [True: 7.01M, False: 4.73M]
  ------------------
 2168|  7.01M|            if (decode_b(t, bl, b[0], PARTITION_NONE, node->o))
  ------------------
  |  Branch (2168:17): [True: 1.95k, False: 7.01M]
  ------------------
 2169|  1.95k|                return -1;
 2170|  7.01M|            break;
 2171|  7.01M|        case PARTITION_H:
  ------------------
  |  Branch (2171:9): [True: 902k, False: 10.8M]
  ------------------
 2172|   902k|            if (decode_b(t, bl, b[0], PARTITION_H, node->h[0]))
  ------------------
  |  Branch (2172:17): [True: 553, False: 901k]
  ------------------
 2173|    553|                return -1;
 2174|   901k|            t->by += hsz;
 2175|   901k|            if (decode_b(t, bl, b[0], PARTITION_H, node->h[1]))
  ------------------
  |  Branch (2175:17): [True: 444, False: 901k]
  ------------------
 2176|    444|                return -1;
 2177|   901k|            t->by -= hsz;
 2178|   901k|            break;
 2179|   617k|        case PARTITION_V:
  ------------------
  |  Branch (2179:9): [True: 617k, False: 11.1M]
  ------------------
 2180|   617k|            if (decode_b(t, bl, b[0], PARTITION_V, node->v[0]))
  ------------------
  |  Branch (2180:17): [True: 1.65k, False: 616k]
  ------------------
 2181|  1.65k|                return -1;
 2182|   616k|            t->bx += hsz;
 2183|   616k|            if (decode_b(t, bl, b[0], PARTITION_V, node->v[1]))
  ------------------
  |  Branch (2183:17): [True: 274, False: 616k]
  ------------------
 2184|    274|                return -1;
 2185|   616k|            t->bx -= hsz;
 2186|   616k|            break;
 2187|  1.88M|        case PARTITION_SPLIT:
  ------------------
  |  Branch (2187:9): [True: 1.88M, False: 9.86M]
  ------------------
 2188|  1.88M|            if (bl == BL_8X8) {
  ------------------
  |  Branch (2188:17): [True: 286k, False: 1.59M]
  ------------------
 2189|   286k|                const EdgeTip *const tip = (const EdgeTip *) node;
 2190|   286k|                assert(hsz == 1);
  ------------------
  |  Branch (2190:17): [True: 286k, False: 1]
  ------------------
 2191|   286k|                if (decode_b(t, bl, BS_4x4, PARTITION_SPLIT, EDGE_ALL_TR_AND_BL))
  ------------------
  |  Branch (2191:21): [True: 292, False: 286k]
  ------------------
 2192|    292|                    return -1;
 2193|   286k|                const enum Filter2d tl_filter = t->tl_4x4_filter;
 2194|   286k|                t->bx++;
 2195|   286k|                if (decode_b(t, bl, BS_4x4, PARTITION_SPLIT, tip->split[0]))
  ------------------
  |  Branch (2195:21): [True: 382, False: 285k]
  ------------------
 2196|    382|                    return -1;
 2197|   285k|                t->bx--;
 2198|   285k|                t->by++;
 2199|   285k|                if (decode_b(t, bl, BS_4x4, PARTITION_SPLIT, tip->split[1]))
  ------------------
  |  Branch (2199:21): [True: 162, False: 285k]
  ------------------
 2200|    162|                    return -1;
 2201|   285k|                t->bx++;
 2202|   285k|                t->tl_4x4_filter = tl_filter;
 2203|   285k|                if (decode_b(t, bl, BS_4x4, PARTITION_SPLIT, tip->split[2]))
  ------------------
  |  Branch (2203:21): [True: 74, False: 285k]
  ------------------
 2204|     74|                    return -1;
 2205|   285k|                t->bx--;
 2206|   285k|                t->by--;
 2207|   285k|#if ARCH_X86_64
 2208|   285k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (2208:21): [True: 285k, False: 18.4E]
  ------------------
 2209|       |                    /* In 8-bit mode with 2-pass decoding the coefficient buffer
 2210|       |                     * can end up misaligned due to skips here. Work around
 2211|       |                     * the issue by explicitly realigning the buffer. */
 2212|   285k|                    const int p = t->frame_thread.pass & 1;
 2213|   285k|                    ts->frame_thread[p].cf =
 2214|   285k|                        (void*)(((uintptr_t)ts->frame_thread[p].cf + 63) & ~63);
 2215|   285k|                }
 2216|   285k|#endif
 2217|  1.59M|            } else {
 2218|  1.59M|                if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 0)))
  ------------------
  |  |   51|  1.59M|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2218:21): [True: 2.95k, False: 1.59M]
  ------------------
 2219|  2.95k|                    return 1;
 2220|  1.59M|                t->bx += hsz;
 2221|  1.59M|                if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 1)))
  ------------------
  |  |   51|  1.59M|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2221:21): [True: 2.87k, False: 1.59M]
  ------------------
 2222|  2.87k|                    return 1;
 2223|  1.59M|                t->bx -= hsz;
 2224|  1.59M|                t->by += hsz;
 2225|  1.59M|                if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 2)))
  ------------------
  |  |   51|  1.59M|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2225:21): [True: 2.20k, False: 1.58M]
  ------------------
 2226|  2.20k|                    return 1;
 2227|  1.58M|                t->bx += hsz;
 2228|  1.58M|                if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 3)))
  ------------------
  |  |   51|  1.58M|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2228:21): [True: 3.14k, False: 1.58M]
  ------------------
 2229|  3.14k|                    return 1;
 2230|  1.58M|                t->bx -= hsz;
 2231|  1.58M|                t->by -= hsz;
 2232|  1.58M|            }
 2233|  1.87M|            break;
 2234|  1.87M|        case PARTITION_T_TOP_SPLIT: {
  ------------------
  |  Branch (2234:9): [True: 145k, False: 11.6M]
  ------------------
 2235|   145k|            if (decode_b(t, bl, b[0], PARTITION_T_TOP_SPLIT, EDGE_ALL_TR_AND_BL))
  ------------------
  |  Branch (2235:17): [True: 167, False: 145k]
  ------------------
 2236|    167|                return -1;
 2237|   145k|            t->bx += hsz;
 2238|   145k|            if (decode_b(t, bl, b[0], PARTITION_T_TOP_SPLIT, node->v[1]))
  ------------------
  |  Branch (2238:17): [True: 238, False: 145k]
  ------------------
 2239|    238|                return -1;
 2240|   145k|            t->bx -= hsz;
 2241|   145k|            t->by += hsz;
 2242|   145k|            if (decode_b(t, bl, b[1], PARTITION_T_TOP_SPLIT, node->h[1]))
  ------------------
  |  Branch (2242:17): [True: 436, False: 145k]
  ------------------
 2243|    436|                return -1;
 2244|   145k|            t->by -= hsz;
 2245|   145k|            break;
 2246|   145k|        }
 2247|   154k|        case PARTITION_T_BOTTOM_SPLIT: {
  ------------------
  |  Branch (2247:9): [True: 154k, False: 11.5M]
  ------------------
 2248|   154k|            if (decode_b(t, bl, b[0], PARTITION_T_BOTTOM_SPLIT, node->h[0]))
  ------------------
  |  Branch (2248:17): [True: 74, False: 154k]
  ------------------
 2249|     74|                return -1;
 2250|   154k|            t->by += hsz;
 2251|   154k|            if (decode_b(t, bl, b[1], PARTITION_T_BOTTOM_SPLIT, node->v[0]))
  ------------------
  |  Branch (2251:17): [True: 79, False: 154k]
  ------------------
 2252|     79|                return -1;
 2253|   154k|            t->bx += hsz;
 2254|   154k|            if (decode_b(t, bl, b[1], PARTITION_T_BOTTOM_SPLIT, 0))
  ------------------
  |  Branch (2254:17): [True: 97, False: 154k]
  ------------------
 2255|     97|                return -1;
 2256|   154k|            t->bx -= hsz;
 2257|   154k|            t->by -= hsz;
 2258|   154k|            break;
 2259|   154k|        }
 2260|   100k|        case PARTITION_T_LEFT_SPLIT: {
  ------------------
  |  Branch (2260:9): [True: 100k, False: 11.6M]
  ------------------
 2261|   100k|            if (decode_b(t, bl, b[0], PARTITION_T_LEFT_SPLIT, EDGE_ALL_TR_AND_BL))
  ------------------
  |  Branch (2261:17): [True: 498, False: 99.6k]
  ------------------
 2262|    498|                return -1;
 2263|  99.6k|            t->by += hsz;
 2264|  99.6k|            if (decode_b(t, bl, b[0], PARTITION_T_LEFT_SPLIT, node->h[1]))
  ------------------
  |  Branch (2264:17): [True: 227, False: 99.4k]
  ------------------
 2265|    227|                return -1;
 2266|  99.4k|            t->by -= hsz;
 2267|  99.4k|            t->bx += hsz;
 2268|  99.4k|            if (decode_b(t, bl, b[1], PARTITION_T_LEFT_SPLIT, node->v[1]))
  ------------------
  |  Branch (2268:17): [True: 318, False: 99.1k]
  ------------------
 2269|    318|                return -1;
 2270|  99.1k|            t->bx -= hsz;
 2271|  99.1k|            break;
 2272|  99.4k|        }
 2273|   281k|        case PARTITION_T_RIGHT_SPLIT: {
  ------------------
  |  Branch (2273:9): [True: 281k, False: 11.4M]
  ------------------
 2274|   281k|            if (decode_b(t, bl, b[0], PARTITION_T_RIGHT_SPLIT, node->v[0]))
  ------------------
  |  Branch (2274:17): [True: 588, False: 280k]
  ------------------
 2275|    588|                return -1;
 2276|   280k|            t->bx += hsz;
 2277|   280k|            if (decode_b(t, bl, b[1], PARTITION_T_RIGHT_SPLIT, node->h[0]))
  ------------------
  |  Branch (2277:17): [True: 341, False: 280k]
  ------------------
 2278|    341|                return -1;
 2279|   280k|            t->by += hsz;
 2280|   280k|            if (decode_b(t, bl, b[1], PARTITION_T_RIGHT_SPLIT, 0))
  ------------------
  |  Branch (2280:17): [True: 210, False: 279k]
  ------------------
 2281|    210|                return -1;
 2282|   279k|            t->by -= hsz;
 2283|   279k|            t->bx -= hsz;
 2284|   279k|            break;
 2285|   280k|        }
 2286|   352k|        case PARTITION_H4: {
  ------------------
  |  Branch (2286:9): [True: 352k, False: 11.3M]
  ------------------
 2287|   352k|            const EdgeBranch *const branch = (const EdgeBranch *) node;
 2288|   352k|            if (decode_b(t, bl, b[0], PARTITION_H4, node->h[0]))
  ------------------
  |  Branch (2288:17): [True: 345, False: 352k]
  ------------------
 2289|    345|                return -1;
 2290|   352k|            t->by += hsz >> 1;
 2291|   352k|            if (decode_b(t, bl, b[0], PARTITION_H4, branch->h4))
  ------------------
  |  Branch (2291:17): [True: 206, False: 352k]
  ------------------
 2292|    206|                return -1;
 2293|   352k|            t->by += hsz >> 1;
 2294|   352k|            if (decode_b(t, bl, b[0], PARTITION_H4, EDGE_ALL_LEFT_HAS_BOTTOM))
  ------------------
  |  Branch (2294:17): [True: 341, False: 351k]
  ------------------
 2295|    341|                return -1;
 2296|   351k|            t->by += hsz >> 1;
 2297|   351k|            if (t->by < f->bh)
  ------------------
  |  Branch (2297:17): [True: 343k, False: 8.73k]
  ------------------
 2298|   343k|                if (decode_b(t, bl, b[0], PARTITION_H4, node->h[1]))
  ------------------
  |  Branch (2298:21): [True: 185, False: 342k]
  ------------------
 2299|    185|                    return -1;
 2300|   351k|            t->by -= hsz * 3 >> 1;
 2301|   351k|            break;
 2302|   351k|        }
 2303|   317k|        case PARTITION_V4: {
  ------------------
  |  Branch (2303:9): [True: 317k, False: 11.4M]
  ------------------
 2304|   317k|            const EdgeBranch *const branch = (const EdgeBranch *) node;
 2305|   317k|            if (decode_b(t, bl, b[0], PARTITION_V4, node->v[0]))
  ------------------
  |  Branch (2305:17): [True: 407, False: 317k]
  ------------------
 2306|    407|                return -1;
 2307|   317k|            t->bx += hsz >> 1;
 2308|   317k|            if (decode_b(t, bl, b[0], PARTITION_V4, branch->v4))
  ------------------
  |  Branch (2308:17): [True: 797, False: 316k]
  ------------------
 2309|    797|                return -1;
 2310|   316k|            t->bx += hsz >> 1;
 2311|   316k|            if (decode_b(t, bl, b[0], PARTITION_V4, EDGE_ALL_TOP_HAS_RIGHT))
  ------------------
  |  Branch (2311:17): [True: 86, False: 316k]
  ------------------
 2312|     86|                return -1;
 2313|   316k|            t->bx += hsz >> 1;
 2314|   316k|            if (t->bx < f->bw)
  ------------------
  |  Branch (2314:17): [True: 305k, False: 10.3k]
  ------------------
 2315|   305k|                if (decode_b(t, bl, b[0], PARTITION_V4, node->v[1]))
  ------------------
  |  Branch (2315:21): [True: 523, False: 305k]
  ------------------
 2316|    523|                    return -1;
 2317|   315k|            t->bx -= hsz * 3 >> 1;
 2318|   315k|            break;
 2319|   316k|        }
 2320|      0|        default: assert(0);
  ------------------
  |  Branch (2320:9): [True: 0, False: 11.7M]
  |  Branch (2320:18): [Folded, False: 0]
  ------------------
 2321|  11.7M|        }
 2322|  11.7M|    } else if (have_h_split) {
  ------------------
  |  Branch (2322:16): [True: 418k, False: 580k]
  ------------------
 2323|   418k|        unsigned is_split;
 2324|   418k|        if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (2324:13): [True: 56.2k, False: 362k]
  ------------------
 2325|  56.2k|            const Av1Block *const b = &f->frame_thread.b[t->by * f->b4_stride + t->bx];
 2326|  56.2k|            is_split = b->bl != bl;
 2327|   362k|        } else {
 2328|   362k|            is_split = dav1d_msac_decode_bool(&ts->msac,
  ------------------
  |  |   54|   362k|#define dav1d_msac_decode_bool           dav1d_msac_decode_bool_sse2
  ------------------
 2329|   362k|                           gather_top_partition_prob(pc, bl));
 2330|   362k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   362k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 362k]
  |  |  ------------------
  |  |   35|   362k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   362k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2331|      0|                printf("poc=%d,y=%d,x=%d,bl=%d,ctx=%d,bp=%d: r=%d\n",
 2332|      0|                       f->frame_hdr->frame_offset, t->by, t->bx, bl, ctx,
 2333|      0|                       is_split ? PARTITION_SPLIT : PARTITION_H, ts->msac.rng);
  ------------------
  |  Branch (2333:24): [True: 0, False: 0]
  ------------------
 2334|   362k|        }
 2335|       |
 2336|   418k|        assert(bl < BL_8X8);
  ------------------
  |  Branch (2336:9): [True: 418k, False: 18.4E]
  ------------------
 2337|   418k|        if (is_split) {
  ------------------
  |  Branch (2337:13): [True: 266k, False: 152k]
  ------------------
 2338|   266k|            bp = PARTITION_SPLIT;
 2339|   266k|            if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 0))) return 1;
  ------------------
  |  |   51|   266k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2339:17): [True: 2.39k, False: 264k]
  ------------------
 2340|   264k|            t->bx += hsz;
 2341|   264k|            if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 1))) return 1;
  ------------------
  |  |   51|   264k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2341:17): [True: 1.63k, False: 262k]
  ------------------
 2342|   262k|            t->bx -= hsz;
 2343|   262k|        } else {
 2344|   152k|            bp = PARTITION_H;
 2345|   152k|            if (decode_b(t, bl, dav1d_block_sizes[bl][PARTITION_H][0],
  ------------------
  |  Branch (2345:17): [True: 317, False: 152k]
  ------------------
 2346|   152k|                         PARTITION_H, node->h[0]))
 2347|    317|                return -1;
 2348|   152k|        }
 2349|   580k|    } else {
 2350|   580k|        assert(have_v_split);
  ------------------
  |  Branch (2350:9): [True: 585k, False: 18.4E]
  ------------------
 2351|   585k|        unsigned is_split;
 2352|   585k|        if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (2352:13): [True: 214k, False: 371k]
  ------------------
 2353|   214k|            const Av1Block *const b = &f->frame_thread.b[t->by * f->b4_stride + t->bx];
 2354|   214k|            is_split = b->bl != bl;
 2355|   371k|        } else {
 2356|   371k|            is_split = dav1d_msac_decode_bool(&ts->msac,
  ------------------
  |  |   54|   371k|#define dav1d_msac_decode_bool           dav1d_msac_decode_bool_sse2
  ------------------
 2357|   371k|                           gather_left_partition_prob(pc, bl));
 2358|   371k|            if (f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I422 && !is_split)
  ------------------
  |  Branch (2358:17): [True: 60.1k, False: 311k]
  |  Branch (2358:63): [True: 620, False: 59.5k]
  ------------------
 2359|    620|                return 1;
 2360|   370k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   370k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 370k]
  |  |  ------------------
  |  |   35|   370k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   370k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2361|      0|                printf("poc=%d,y=%d,x=%d,bl=%d,ctx=%d,bp=%d: r=%d\n",
 2362|      0|                       f->frame_hdr->frame_offset, t->by, t->bx, bl, ctx,
 2363|      0|                       is_split ? PARTITION_SPLIT : PARTITION_V, ts->msac.rng);
  ------------------
  |  Branch (2363:24): [True: 0, False: 0]
  ------------------
 2364|   370k|        }
 2365|       |
 2366|   585k|        assert(bl < BL_8X8);
  ------------------
  |  Branch (2366:9): [True: 585k, False: 18.4E]
  ------------------
 2367|   585k|        if (is_split) {
  ------------------
  |  Branch (2367:13): [True: 208k, False: 377k]
  ------------------
 2368|   208k|            bp = PARTITION_SPLIT;
 2369|   208k|            if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 0))) return 1;
  ------------------
  |  |   51|   208k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2369:17): [True: 65.5k, False: 142k]
  ------------------
 2370|   142k|            t->by += hsz;
 2371|   142k|            if (decode_sb(t, bl + 1, INTRA_EDGE_SPLIT(node, 2))) return 1;
  ------------------
  |  |   51|   142k|    ((const EdgeNode*)((uintptr_t)(n) + ((const EdgeBranch*)(n))->split_offset[i]))
  ------------------
  |  Branch (2371:17): [True: 8.77k, False: 133k]
  ------------------
 2372|   133k|            t->by -= hsz;
 2373|   377k|        } else {
 2374|   377k|            bp = PARTITION_V;
 2375|   377k|            if (decode_b(t, bl, dav1d_block_sizes[bl][PARTITION_V][0],
  ------------------
  |  Branch (2375:17): [True: 132, False: 377k]
  ------------------
 2376|   377k|                         PARTITION_V, node->v[0]))
 2377|    132|                return -1;
 2378|   377k|        }
 2379|   585k|    }
 2380|       |
 2381|  12.6M|    if (t->frame_thread.pass != 2 && (bp != PARTITION_SPLIT || bl == BL_8X8)) {
  ------------------
  |  Branch (2381:9): [True: 8.74M, False: 3.89M]
  |  Branch (2381:39): [True: 6.93M, False: 1.80M]
  |  Branch (2381:64): [True: 235k, False: 1.57M]
  ------------------
 2382|  7.17M|#define set_ctx(rep_macro) \
 2383|  7.17M|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
 2384|  7.17M|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
 2385|  7.17M|        case_set_upto16(ulog2(hsz));
  ------------------
  |  |   80|  7.17M|    switch (var) { \
  |  |   81|  1.37M|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 2383|  1.37M|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|  1.37M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  1.37M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|  1.37M|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|  1.37M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  1.37M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (81:5): [True: 1.37M, False: 5.80M]
  |  |  ------------------
  |  |   82|  1.82M|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 2383|  1.82M|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|  1.82M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.82M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|  1.82M|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|  1.82M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.82M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (82:5): [True: 1.82M, False: 5.34M]
  |  |  ------------------
  |  |   83|   937k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 2383|   937k|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|   937k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   937k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|   937k|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|   937k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   937k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (83:5): [True: 937k, False: 6.23M]
  |  |  ------------------
  |  |   84|  1.81M|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 2383|  1.81M|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  1.81M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.81M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|  1.81M|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  1.81M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.81M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (84:5): [True: 1.81M, False: 5.35M]
  |  |  ------------------
  |  |   85|  1.22M|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 2383|  1.22M|        rep_macro(t->a->partition, bx8, dav1d_al_part_ctx[0][bl][bp]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  1.22M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.22M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.22M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.22M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.22M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 2384|  1.22M|        rep_macro(t->l.partition, by8, dav1d_al_part_ctx[1][bl][bp])
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  1.22M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.22M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.22M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.22M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.22M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (85:5): [True: 1.22M, False: 5.95M]
  |  |  ------------------
  |  |   86|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (86:5): [True: 0, False: 7.17M]
  |  |  ------------------
  |  |   87|  7.17M|    }
  ------------------
  |  Branch (2385:9): [Folded, False: 0]
  ------------------
 2386|  7.17M|#undef set_ctx
 2387|  7.17M|    }
 2388|       |
 2389|  12.6M|    return 0;
 2390|  12.6M|}
decode.c:decode_b:
  687|  16.3M|                    const enum EdgeFlags intra_edge_flags) {
  688|  16.3M|    Dav1dTileState *const ts = t->ts;
  689|  16.3M|    const Dav1dFrameContext *const f = t->f;
  690|  16.3M|    Av1Block b_mem, *const b = t->frame_thread.pass ?
  ------------------
  |  Branch (690:32): [True: 16.3M, False: 18.4E]
  ------------------
  691|  18.4E|        &f->frame_thread.b[t->by * f->b4_stride + t->bx] : &b_mem;
  692|  16.3M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  693|  16.3M|    const int bx4 = t->bx & 31, by4 = t->by & 31;
  694|  16.3M|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  695|  16.3M|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  696|  16.3M|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
  697|  16.3M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
  698|  16.3M|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
  699|  16.3M|    const int cbw4 = (bw4 + ss_hor) >> ss_hor, cbh4 = (bh4 + ss_ver) >> ss_ver;
  700|  16.3M|    const int have_left = t->bx > ts->tiling.col_start;
  701|  16.3M|    const int have_top = t->by > ts->tiling.row_start;
  702|  16.3M|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (702:28): [True: 10.6M, False: 5.73M]
  ------------------
  703|  10.6M|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (703:29): [True: 9.69M, False: 946k]
  |  Branch (703:45): [True: 472k, False: 473k]
  ------------------
  704|  10.1M|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (704:29): [True: 9.29M, False: 881k]
  |  Branch (704:45): [True: 440k, False: 441k]
  ------------------
  705|       |
  706|  16.3M|    if (t->frame_thread.pass == 2) {
  ------------------
  |  Branch (706:9): [True: 4.56M, False: 11.8M]
  ------------------
  707|  4.56M|        if (b->intra) {
  ------------------
  |  Branch (707:13): [True: 1.73M, False: 2.82M]
  ------------------
  708|  1.73M|            f->bd_fn.recon_b_intra(t, bs, intra_edge_flags, b);
  709|       |
  710|  1.73M|            const enum IntraPredMode y_mode_nofilt =
  711|  1.73M|                b->y_mode == FILTER_PRED ? DC_PRED : b->y_mode;
  ------------------
  |  Branch (711:17): [True: 249k, False: 1.48M]
  ------------------
  712|  1.73M|#define set_ctx(rep_macro) \
  713|  1.73M|            rep_macro(edge->mode, off, y_mode_nofilt); \
  714|  1.73M|            rep_macro(edge->intra, off, 1)
  715|  1.73M|            BlockContext *edge = t->a;
  716|  5.20M|            for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (716:40): [True: 3.47M, False: 1.73M]
  ------------------
  717|  3.47M|                case_set(b_dim[2 + i]);
  ------------------
  |  |   70|  3.47M|    switch (var) { \
  |  |   71|   519k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  |  713|   519k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   519k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   519k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|   519k|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   519k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   519k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 519k, False: 2.95M]
  |  |  ------------------
  |  |   72|   712k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  |  713|   712k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   712k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   712k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|   712k|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   712k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   712k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 712k, False: 2.75M]
  |  |  ------------------
  |  |   73|   586k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  |  713|   586k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   586k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   586k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|   586k|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   586k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   586k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 586k, False: 2.88M]
  |  |  ------------------
  |  |   74|   406k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  |  713|   406k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   406k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   406k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|   406k|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   406k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   406k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 406k, False: 3.06M]
  |  |  ------------------
  |  |   75|   810k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  |  713|   810k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   810k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   810k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   810k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   810k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 810k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|   810k|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   810k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   810k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   810k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   810k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 810k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 810k, False: 2.65M]
  |  |  ------------------
  |  |   76|   435k|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  |  713|   435k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   435k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   435k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   435k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   435k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 435k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  714|   435k|            rep_macro(edge->intra, off, 1)
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   435k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   435k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   435k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   435k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 435k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 435k, False: 3.03M]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 3.47M]
  |  |  ------------------
  |  |   78|  3.47M|    }
  ------------------
  |  Branch (717:17): [Folded, False: 0]
  ------------------
  718|  3.47M|            }
  719|  1.73M|#undef set_ctx
  720|  1.73M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|  1.73M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 269k, False: 1.46M]
  |  |  ------------------
  ------------------
  721|   269k|                refmvs_block *const r = &t->rt.r[(t->by & 31) + 5 + bh4 - 1][t->bx];
  722|  1.38M|                for (int x = 0; x < bw4; x++) {
  ------------------
  |  Branch (722:33): [True: 1.11M, False: 269k]
  ------------------
  723|  1.11M|                    r[x].ref.ref[0] = 0;
  724|  1.11M|                    r[x].bs = bs;
  725|  1.11M|                }
  726|   269k|                refmvs_block *const *rr = &t->rt.r[(t->by & 31) + 5];
  727|  1.10M|                for (int y = 0; y < bh4 - 1; y++) {
  ------------------
  |  Branch (727:33): [True: 830k, False: 269k]
  ------------------
  728|   830k|                    rr[y][t->bx + bw4 - 1].ref.ref[0] = 0;
  729|   830k|                    rr[y][t->bx + bw4 - 1].bs = bs;
  730|   830k|                }
  731|   269k|            }
  732|       |
  733|  1.73M|            if (has_chroma) {
  ------------------
  |  Branch (733:17): [True: 976k, False: 758k]
  ------------------
  734|   976k|                uint8_t uv_mode = b->uv_mode;
  735|   976k|                dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], uv_mode);
  736|   976k|                dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], uv_mode);
  737|   976k|            }
  738|  2.82M|        } else {
  739|  2.82M|            if (IS_INTER_OR_SWITCH(f->frame_hdr) /* not intrabc */ &&
  ------------------
  |  |   36|  5.65M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 2.81M, False: 17.5k]
  |  |  ------------------
  ------------------
  740|  2.81M|                b->comp_type == COMP_INTER_NONE && b->motion_mode == MM_WARP)
  ------------------
  |  Branch (740:17): [True: 2.67M, False: 140k]
  |  Branch (740:52): [True: 134k, False: 2.53M]
  ------------------
  741|   134k|            {
  742|   134k|                if (b->matrix[0] == INT16_MIN) {
  ------------------
  |  Branch (742:21): [True: 10.4k, False: 124k]
  ------------------
  743|  10.4k|                    t->warpmv.type = DAV1D_WM_TYPE_IDENTITY;
  744|   124k|                } else {
  745|   124k|                    t->warpmv.type = DAV1D_WM_TYPE_AFFINE;
  746|   124k|                    t->warpmv.matrix[2] = b->matrix[0] + 0x10000;
  747|   124k|                    t->warpmv.matrix[3] = b->matrix[1];
  748|   124k|                    t->warpmv.matrix[4] = b->matrix[2];
  749|   124k|                    t->warpmv.matrix[5] = b->matrix[3] + 0x10000;
  750|   124k|                    dav1d_set_affine_mv2d(bw4, bh4, b->mv2d, &t->warpmv,
  751|   124k|                                          t->bx, t->by);
  752|   124k|                    dav1d_get_shear_params(&t->warpmv);
  753|   124k|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  754|   124k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   124k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 124k]
  |  |  ------------------
  |  |   35|   124k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   124k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  755|      0|                        printf("[ %c%x %c%x %c%x\n  %c%x %c%x %c%x ]\n"
  756|      0|                               "alpha=%c%x, beta=%c%x, gamma=%c%x, delta=%c%x, mv=y:%d,x:%d\n",
  757|      0|                               signabs(t->warpmv.matrix[0]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  758|      0|                               signabs(t->warpmv.matrix[1]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  759|      0|                               signabs(t->warpmv.matrix[2]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  760|      0|                               signabs(t->warpmv.matrix[3]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  761|      0|                               signabs(t->warpmv.matrix[4]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  762|      0|                               signabs(t->warpmv.matrix[5]),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  763|      0|                               signabs(t->warpmv.u.p.alpha),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  764|      0|                               signabs(t->warpmv.u.p.beta),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  765|      0|                               signabs(t->warpmv.u.p.gamma),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  766|      0|                               signabs(t->warpmv.u.p.delta),
  ------------------
  |  |  753|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (753:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  767|      0|                               b->mv2d.y, b->mv2d.x);
  768|   124k|#undef signabs
  769|   124k|                }
  770|   134k|            }
  771|  2.82M|            if (f->bd_fn.recon_b_inter(t, bs, b)) return -1;
  ------------------
  |  Branch (771:17): [True: 0, False: 2.82M]
  ------------------
  772|       |
  773|  2.82M|            const uint8_t *const filter = dav1d_filter_dir[b->filter2d];
  774|  2.82M|            BlockContext *edge = t->a;
  775|  8.48M|            for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (775:40): [True: 5.66M, False: 2.82M]
  ------------------
  776|  5.66M|#define set_ctx(rep_macro) \
  777|  5.66M|                rep_macro(edge->filter[0], off, filter[0]); \
  778|  5.66M|                rep_macro(edge->filter[1], off, filter[1]); \
  779|  5.66M|                rep_macro(edge->intra, off, 0)
  780|  5.66M|                case_set(b_dim[2 + i]);
  ------------------
  |  |   70|  5.66M|    switch (var) { \
  |  |   71|   398k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  |  777|   398k|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   398k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   398k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|   398k|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   398k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   398k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|   398k|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   398k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   398k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 398k, False: 5.26M]
  |  |  ------------------
  |  |   72|   831k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  |  777|   831k|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   831k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   831k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|   831k|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   831k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   831k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|   831k|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   831k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   831k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 831k, False: 4.82M]
  |  |  ------------------
  |  |   73|   679k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  |  777|   679k|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   679k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   679k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|   679k|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   679k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   679k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|   679k|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   679k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   679k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 679k, False: 4.98M]
  |  |  ------------------
  |  |   74|   257k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  |  777|   257k|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   257k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   257k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|   257k|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   257k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   257k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|   257k|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   257k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   257k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 257k, False: 5.40M]
  |  |  ------------------
  |  |   75|  2.11M|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  |  777|  2.11M|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.11M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.11M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.11M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.11M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.11M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|  2.11M|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.11M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.11M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.11M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.11M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.11M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|  2.11M|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.11M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.11M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.11M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.11M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.11M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 2.11M, False: 3.54M]
  |  |  ------------------
  |  |   76|  1.37M|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  |  777|  1.37M|                rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.37M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.37M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.37M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.37M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.37M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  778|  1.37M|                rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.37M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.37M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.37M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.37M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.37M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  779|  1.37M|                rep_macro(edge->intra, off, 0)
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.37M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.37M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.37M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.37M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.37M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 1.37M, False: 4.28M]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 5.66M]
  |  |  ------------------
  |  |   78|  5.66M|    }
  ------------------
  |  Branch (780:17): [Folded, False: 0]
  ------------------
  781|  5.66M|#undef set_ctx
  782|  5.66M|            }
  783|       |
  784|  2.82M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|  2.82M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 2.81M, False: 17.9k]
  |  |  ------------------
  ------------------
  785|  2.81M|                refmvs_block *const r = &t->rt.r[(t->by & 31) + 5 + bh4 - 1][t->bx];
  786|  2.81M|                const int ref1 = b->ref[0] + 1;
  787|  2.81M|                const union mv mv1 = b->mv[0];
  788|  44.7M|                for (int x = 0; x < bw4; x++) {
  ------------------
  |  Branch (788:33): [True: 41.9M, False: 2.81M]
  ------------------
  789|  41.9M|                    r[x].ref.ref[0] = ref1;
  790|  41.9M|                    r[x].mv.mv[0] = mv1;
  791|  41.9M|                    r[x].bs = bs;
  792|  41.9M|                }
  793|  2.81M|                refmvs_block *const *rr = &t->rt.r[(t->by & 31) + 5];
  794|  42.2M|                for (int y = 0; y < bh4 - 1; y++) {
  ------------------
  |  Branch (794:33): [True: 39.4M, False: 2.81M]
  ------------------
  795|  39.4M|                    rr[y][t->bx + bw4 - 1].ref.ref[0] = ref1;
  796|  39.4M|                    rr[y][t->bx + bw4 - 1].mv.mv[0] = mv1;
  797|  39.4M|                    rr[y][t->bx + bw4 - 1].bs = bs;
  798|  39.4M|                }
  799|  2.81M|            }
  800|       |
  801|  2.82M|            if (has_chroma) {
  ------------------
  |  Branch (801:17): [True: 924k, False: 1.90M]
  ------------------
  802|   924k|                dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], DC_PRED);
  803|   924k|                dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], DC_PRED);
  804|   924k|            }
  805|  2.82M|        }
  806|  4.56M|        return 0;
  807|  4.56M|    }
  808|       |
  809|  11.8M|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
  810|       |
  811|  11.8M|    b->bl = bl;
  812|  11.8M|    b->bp = bp;
  813|  11.8M|    b->bs = bs;
  814|       |
  815|  11.8M|    const Dav1dSegmentationData *seg = NULL;
  816|       |
  817|       |    // segment_id (if seg_feature for skip/ref/gmv is enabled)
  818|  11.8M|    int seg_pred = 0;
  819|  11.8M|    if (f->frame_hdr->segmentation.enabled) {
  ------------------
  |  Branch (819:9): [True: 3.46M, False: 8.34M]
  ------------------
  820|  3.46M|        if (!f->frame_hdr->segmentation.update_map) {
  ------------------
  |  Branch (820:13): [True: 906k, False: 2.55M]
  ------------------
  821|   906k|            if (f->prev_segmap) {
  ------------------
  |  Branch (821:17): [True: 695k, False: 210k]
  ------------------
  822|   695k|                unsigned seg_id = get_prev_frame_segid(f, t->by, t->bx, w4, h4,
  823|   695k|                                                       f->prev_segmap,
  824|   695k|                                                       f->b4_stride);
  825|   695k|                if (seg_id >= 8) return -1;
  ------------------
  |  Branch (825:21): [True: 0, False: 695k]
  ------------------
  826|   695k|                b->seg_id = seg_id;
  827|   695k|            } else {
  828|   210k|                b->seg_id = 0;
  829|   210k|            }
  830|   906k|            seg = &f->frame_hdr->segmentation.seg_data.d[b->seg_id];
  831|  2.55M|        } else if (f->frame_hdr->segmentation.seg_data.preskip) {
  ------------------
  |  Branch (831:20): [True: 2.15M, False: 408k]
  ------------------
  832|  2.15M|            if (f->frame_hdr->segmentation.temporal &&
  ------------------
  |  Branch (832:17): [True: 794k, False: 1.35M]
  ------------------
  833|   794k|                (seg_pred = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   794k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (833:17): [True: 90.5k, False: 703k]
  ------------------
  834|   794k|                                ts->cdf.m.seg_pred[t->a->seg_pred[bx4] +
  835|   794k|                                t->l.seg_pred[by4]])))
  836|  90.5k|            {
  837|       |                // temporal predicted seg_id
  838|  90.5k|                if (f->prev_segmap) {
  ------------------
  |  Branch (838:21): [True: 65.9k, False: 24.5k]
  ------------------
  839|  65.9k|                    unsigned seg_id = get_prev_frame_segid(f, t->by, t->bx,
  840|  65.9k|                                                           w4, h4,
  841|  65.9k|                                                           f->prev_segmap,
  842|  65.9k|                                                           f->b4_stride);
  843|  65.9k|                    if (seg_id >= 8) return -1;
  ------------------
  |  Branch (843:25): [True: 0, False: 65.9k]
  ------------------
  844|  65.9k|                    b->seg_id = seg_id;
  845|  65.9k|                } else {
  846|  24.5k|                    b->seg_id = 0;
  847|  24.5k|                }
  848|  2.06M|            } else {
  849|  2.06M|                int seg_ctx;
  850|  2.06M|                const unsigned pred_seg_id =
  851|  2.06M|                    get_cur_frame_segid(t->by, t->bx, have_top, have_left,
  852|  2.06M|                                        &seg_ctx, f->cur_segmap, f->b4_stride);
  853|  2.06M|                const unsigned diff = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  2.06M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  854|  2.06M|                                          ts->cdf.m.seg_id[seg_ctx],
  855|  2.06M|                                          DAV1D_MAX_SEGMENTS - 1);
  ------------------
  |  |   43|  2.06M|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  856|  2.06M|                const unsigned last_active_seg_id =
  857|  2.06M|                    f->frame_hdr->segmentation.seg_data.last_active_segid;
  858|  2.06M|                b->seg_id = neg_deinterleave(diff, pred_seg_id,
  859|  2.06M|                                             last_active_seg_id + 1);
  860|  2.06M|                if (b->seg_id > last_active_seg_id) b->seg_id = 0; // error?
  ------------------
  |  Branch (860:21): [True: 106k, False: 1.95M]
  ------------------
  861|  2.06M|                if (b->seg_id >= DAV1D_MAX_SEGMENTS) b->seg_id = 0; // error?
  ------------------
  |  |   43|  2.06M|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (861:21): [True: 0, False: 2.06M]
  ------------------
  862|  2.06M|            }
  863|       |
  864|  2.15M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  2.15M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 2.15M]
  |  |  ------------------
  |  |   35|  2.15M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  2.15M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  865|      0|                printf("Post-segid[preskip;%d]: r=%d\n",
  866|      0|                       b->seg_id, ts->msac.rng);
  867|       |
  868|  2.15M|            seg = &f->frame_hdr->segmentation.seg_data.d[b->seg_id];
  869|  2.15M|        }
  870|  8.34M|    } else {
  871|  8.34M|        b->seg_id = 0;
  872|  8.34M|    }
  873|       |
  874|       |    // skip_mode
  875|  11.8M|    if ((!seg || (!seg->globalmv && seg->ref == -1 && !seg->skip)) &&
  ------------------
  |  Branch (875:10): [True: 8.75M, False: 3.05M]
  |  Branch (875:19): [True: 654k, False: 2.40M]
  |  Branch (875:37): [True: 356k, False: 297k]
  |  Branch (875:55): [True: 293k, False: 62.8k]
  ------------------
  876|  9.01M|        f->frame_hdr->skip_mode_enabled && imin(bw4, bh4) > 1)
  ------------------
  |  Branch (876:9): [True: 242k, False: 8.77M]
  |  Branch (876:44): [True: 207k, False: 35.3k]
  ------------------
  877|   207k|    {
  878|   207k|        const int smctx = t->a->skip_mode[bx4] + t->l.skip_mode[by4];
  879|   207k|        b->skip_mode = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   207k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  880|   207k|                           ts->cdf.m.skip_mode[smctx]);
  881|   207k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   207k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 207k]
  |  |  ------------------
  |  |   35|   207k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   207k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  882|      0|            printf("Post-skipmode[%d]: r=%d\n", b->skip_mode, ts->msac.rng);
  883|  11.6M|    } else {
  884|  11.6M|        b->skip_mode = 0;
  885|  11.6M|    }
  886|       |
  887|       |    // skip
  888|  11.8M|    if (b->skip_mode || (seg && seg->skip)) {
  ------------------
  |  Branch (888:9): [True: 61.4k, False: 11.7M]
  |  Branch (888:26): [True: 3.05M, False: 8.69M]
  |  Branch (888:33): [True: 2.54M, False: 507k]
  ------------------
  889|  2.57M|        b->skip = 1;
  890|  9.24M|    } else {
  891|  9.24M|        const int sctx = t->a->skip[bx4] + t->l.skip[by4];
  892|  9.24M|        b->skip = dav1d_msac_decode_bool_adapt(&ts->msac, ts->cdf.m.skip[sctx]);
  ------------------
  |  |   52|  9.24M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  893|  9.24M|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  9.24M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 9.24M]
  |  |  ------------------
  |  |   35|  9.24M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  9.24M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  894|      0|            printf("Post-skip[%d]: r=%d\n", b->skip, ts->msac.rng);
  895|  9.24M|    }
  896|       |
  897|       |    // segment_id
  898|  11.8M|    if (f->frame_hdr->segmentation.enabled &&
  ------------------
  |  Branch (898:9): [True: 3.46M, False: 8.34M]
  ------------------
  899|  3.46M|        f->frame_hdr->segmentation.update_map &&
  ------------------
  |  Branch (899:9): [True: 2.56M, False: 903k]
  ------------------
  900|  2.56M|        !f->frame_hdr->segmentation.seg_data.preskip)
  ------------------
  |  Branch (900:9): [True: 408k, False: 2.15M]
  ------------------
  901|   408k|    {
  902|   408k|        if (!b->skip && f->frame_hdr->segmentation.temporal &&
  ------------------
  |  Branch (902:13): [True: 265k, False: 143k]
  |  Branch (902:25): [True: 15.3k, False: 250k]
  ------------------
  903|  15.3k|            (seg_pred = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  15.3k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (903:13): [True: 7.21k, False: 8.08k]
  ------------------
  904|  15.3k|                            ts->cdf.m.seg_pred[t->a->seg_pred[bx4] +
  905|  15.3k|                            t->l.seg_pred[by4]])))
  906|  7.21k|        {
  907|       |            // temporal predicted seg_id
  908|  7.21k|            if (f->prev_segmap) {
  ------------------
  |  Branch (908:17): [True: 1.39k, False: 5.82k]
  ------------------
  909|  1.39k|                unsigned seg_id = get_prev_frame_segid(f, t->by, t->bx, w4, h4,
  910|  1.39k|                                                       f->prev_segmap,
  911|  1.39k|                                                       f->b4_stride);
  912|  1.39k|                if (seg_id >= 8) return -1;
  ------------------
  |  Branch (912:21): [True: 0, False: 1.39k]
  ------------------
  913|  1.39k|                b->seg_id = seg_id;
  914|  5.82k|            } else {
  915|  5.82k|                b->seg_id = 0;
  916|  5.82k|            }
  917|   401k|        } else {
  918|   401k|            int seg_ctx;
  919|   401k|            const unsigned pred_seg_id =
  920|   401k|                get_cur_frame_segid(t->by, t->bx, have_top, have_left,
  921|   401k|                                    &seg_ctx, f->cur_segmap, f->b4_stride);
  922|   401k|            if (b->skip) {
  ------------------
  |  Branch (922:17): [True: 145k, False: 255k]
  ------------------
  923|   145k|                b->seg_id = pred_seg_id;
  924|   255k|            } else {
  925|   255k|                const unsigned diff = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   255k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  926|   255k|                                          ts->cdf.m.seg_id[seg_ctx],
  927|   255k|                                          DAV1D_MAX_SEGMENTS - 1);
  ------------------
  |  |   43|   255k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  928|   255k|                const unsigned last_active_seg_id =
  929|   255k|                    f->frame_hdr->segmentation.seg_data.last_active_segid;
  930|   255k|                b->seg_id = neg_deinterleave(diff, pred_seg_id,
  931|   255k|                                             last_active_seg_id + 1);
  932|   255k|                if (b->seg_id > last_active_seg_id) b->seg_id = 0; // error?
  ------------------
  |  Branch (932:21): [True: 9.37k, False: 246k]
  ------------------
  933|   255k|            }
  934|   401k|            if (b->seg_id >= DAV1D_MAX_SEGMENTS) b->seg_id = 0; // error?
  ------------------
  |  |   43|   401k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (934:17): [True: 1.56k, False: 399k]
  ------------------
  935|   401k|        }
  936|       |
  937|   408k|        seg = &f->frame_hdr->segmentation.seg_data.d[b->seg_id];
  938|       |
  939|   408k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   408k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 408k]
  |  |  ------------------
  |  |   35|   408k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   408k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  940|      0|            printf("Post-segid[postskip;%d]: r=%d\n",
  941|      0|                   b->seg_id, ts->msac.rng);
  942|   408k|    }
  943|       |
  944|       |    // cdef index
  945|  11.8M|    if (!b->skip) {
  ------------------
  |  Branch (945:9): [True: 6.59M, False: 5.22M]
  ------------------
  946|  6.59M|        const int idx = f->seq_hdr->sb128 ? ((t->bx & 16) >> 4) +
  ------------------
  |  Branch (946:25): [True: 5.06M, False: 1.52M]
  ------------------
  947|  5.06M|                                           ((t->by & 16) >> 3) : 0;
  948|  6.59M|        if (t->cur_sb_cdef_idx_ptr[idx] == -1) {
  ------------------
  |  Branch (948:13): [True: 1.04M, False: 5.54M]
  ------------------
  949|  1.04M|            const int v = dav1d_msac_decode_bools(&ts->msac,
  950|  1.04M|                              f->frame_hdr->cdef.n_bits);
  951|  1.04M|            t->cur_sb_cdef_idx_ptr[idx] = v;
  952|  1.04M|            if (bw4 > 16) t->cur_sb_cdef_idx_ptr[idx + 1] = v;
  ------------------
  |  Branch (952:17): [True: 191k, False: 853k]
  ------------------
  953|  1.04M|            if (bh4 > 16) t->cur_sb_cdef_idx_ptr[idx + 2] = v;
  ------------------
  |  Branch (953:17): [True: 192k, False: 852k]
  ------------------
  954|  1.04M|            if (bw4 == 32 && bh4 == 32) t->cur_sb_cdef_idx_ptr[idx + 3] = v;
  ------------------
  |  Branch (954:17): [True: 191k, False: 853k]
  |  Branch (954:30): [True: 178k, False: 12.8k]
  ------------------
  955|       |
  956|  1.04M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.04M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.04M]
  |  |  ------------------
  |  |   35|  1.04M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.04M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  957|      0|                printf("Post-cdef_idx[%d]: r=%d\n",
  958|      0|                        *t->cur_sb_cdef_idx_ptr, ts->msac.rng);
  959|  1.04M|        }
  960|  6.59M|    }
  961|       |
  962|       |    // delta-q/lf
  963|  11.8M|    if (!((t->bx | t->by) & (31 >> !f->seq_hdr->sb128))) {
  ------------------
  |  Branch (963:9): [True: 3.11M, False: 8.70M]
  ------------------
  964|  3.11M|        const int prev_qidx = ts->last_qidx;
  965|  3.11M|        const int have_delta_q = f->frame_hdr->delta.q.present &&
  ------------------
  |  Branch (965:34): [True: 1.26M, False: 1.84M]
  ------------------
  966|  1.26M|            (bs != (f->seq_hdr->sb128 ? BS_128x128 : BS_64x64) || !b->skip);
  ------------------
  |  Branch (966:14): [True: 148k, False: 1.11M]
  |  Branch (966:21): [True: 118k, False: 1.14M]
  |  Branch (966:67): [True: 23.9k, False: 1.09M]
  ------------------
  967|       |
  968|  3.11M|        uint32_t prev_delta_lf = ts->last_delta_lf.u32;
  969|       |
  970|  3.11M|        if (have_delta_q) {
  ------------------
  |  Branch (970:13): [True: 172k, False: 2.93M]
  ------------------
  971|   172k|            int delta_q = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   172k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  972|   172k|                                                          ts->cdf.m.delta_q, 3);
  973|   172k|            if (delta_q == 3) {
  ------------------
  |  Branch (973:17): [True: 30.7k, False: 141k]
  ------------------
  974|  30.7k|                const int n_bits = 1 + dav1d_msac_decode_bools(&ts->msac, 3);
  975|  30.7k|                delta_q = dav1d_msac_decode_bools(&ts->msac, n_bits) +
  976|  30.7k|                          1 + (1 << n_bits);
  977|  30.7k|            }
  978|   172k|            if (delta_q) {
  ------------------
  |  Branch (978:17): [True: 51.1k, False: 121k]
  ------------------
  979|  51.1k|                if (dav1d_msac_decode_bool_equi(&ts->msac)) delta_q = -delta_q;
  ------------------
  |  |   53|  51.1k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (979:21): [True: 38.7k, False: 12.4k]
  ------------------
  980|  51.1k|                delta_q *= 1 << f->frame_hdr->delta.q.res_log2;
  981|  51.1k|            }
  982|   172k|            ts->last_qidx = iclip(ts->last_qidx + delta_q, 1, 255);
  983|   172k|            if (have_delta_q && DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   172k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 172k]
  |  |  ------------------
  |  |   35|   172k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   172k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (983:17): [True: 172k, False: 18.4E]
  ------------------
  984|      0|                printf("Post-delta_q[%d->%d]: r=%d\n",
  985|      0|                       delta_q, ts->last_qidx, ts->msac.rng);
  986|       |
  987|   172k|            if (f->frame_hdr->delta.lf.present) {
  ------------------
  |  Branch (987:17): [True: 62.9k, False: 109k]
  ------------------
  988|  62.9k|                const int n_lfs = f->frame_hdr->delta.lf.multi ?
  ------------------
  |  Branch (988:35): [True: 46.3k, False: 16.5k]
  ------------------
  989|  46.3k|                    f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 ? 4 : 2 : 1;
  ------------------
  |  Branch (989:21): [True: 39.1k, False: 7.22k]
  ------------------
  990|       |
  991|   250k|                for (int i = 0; i < n_lfs; i++) {
  ------------------
  |  Branch (991:33): [True: 187k, False: 62.9k]
  ------------------
  992|   187k|                    int delta_lf = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   187k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  993|   187k|                        ts->cdf.m.delta_lf[i + f->frame_hdr->delta.lf.multi], 3);
  994|   187k|                    if (delta_lf == 3) {
  ------------------
  |  Branch (994:25): [True: 48.8k, False: 138k]
  ------------------
  995|  48.8k|                        const int n_bits = 1 + dav1d_msac_decode_bools(&ts->msac, 3);
  996|  48.8k|                        delta_lf = dav1d_msac_decode_bools(&ts->msac, n_bits) +
  997|  48.8k|                                   1 + (1 << n_bits);
  998|  48.8k|                    }
  999|   187k|                    if (delta_lf) {
  ------------------
  |  Branch (999:25): [True: 67.9k, False: 119k]
  ------------------
 1000|  67.9k|                        if (dav1d_msac_decode_bool_equi(&ts->msac))
  ------------------
  |  |   53|  67.9k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (1000:29): [True: 58.3k, False: 9.65k]
  ------------------
 1001|  58.3k|                            delta_lf = -delta_lf;
 1002|  67.9k|                        delta_lf *= 1 << f->frame_hdr->delta.lf.res_log2;
 1003|  67.9k|                    }
 1004|   187k|                    ts->last_delta_lf.i8[i] =
 1005|   187k|                        iclip(ts->last_delta_lf.i8[i] + delta_lf, -63, 63);
 1006|   187k|                    if (have_delta_q && DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   187k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 187k]
  |  |  ------------------
  |  |   35|   187k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   187k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (1006:25): [True: 187k, False: 18.4E]
  ------------------
 1007|      0|                        printf("Post-delta_lf[%d:%d]: r=%d\n", i, delta_lf,
 1008|      0|                               ts->msac.rng);
 1009|   187k|                }
 1010|  62.9k|            }
 1011|   172k|        }
 1012|  3.11M|        if (ts->last_qidx == f->frame_hdr->quant.yac) {
  ------------------
  |  Branch (1012:13): [True: 2.97M, False: 130k]
  ------------------
 1013|       |            // assign frame-wide q values to this sb
 1014|  2.97M|            ts->dq = f->dq;
 1015|  2.97M|        } else if (ts->last_qidx != prev_qidx) {
  ------------------
  |  Branch (1015:20): [True: 20.5k, False: 109k]
  ------------------
 1016|       |            // find sb-specific quant parameters
 1017|  20.5k|            init_quant_tables(f->seq_hdr, f->frame_hdr, ts->last_qidx, ts->dqmem);
 1018|  20.5k|            ts->dq = ts->dqmem;
 1019|  20.5k|        }
 1020|  3.11M|        if (!ts->last_delta_lf.u32) {
  ------------------
  |  Branch (1020:13): [True: 3.06M, False: 49.3k]
  ------------------
 1021|       |            // assign frame-wide lf values to this sb
 1022|  3.06M|            ts->lflvl = f->lf.lvl;
 1023|  3.06M|        } else if (ts->last_delta_lf.u32 != prev_delta_lf) {
  ------------------
  |  Branch (1023:20): [True: 16.3k, False: 32.9k]
  ------------------
 1024|       |            // find sb-specific lf lvl parameters
 1025|  16.3k|            ts->lflvl = ts->lflvlmem;
 1026|  16.3k|            dav1d_calc_lf_values(ts->lflvlmem, f->frame_hdr, ts->last_delta_lf.i8);
 1027|  16.3k|        }
 1028|  3.11M|    }
 1029|       |
 1030|  11.8M|    if (b->skip_mode) {
  ------------------
  |  Branch (1030:9): [True: 25.3k, False: 11.7M]
  ------------------
 1031|  25.3k|        b->intra = 0;
 1032|  11.7M|    } else if (IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|  11.7M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 4.35M, False: 7.43M]
  |  |  ------------------
  ------------------
 1033|  4.35M|        if (seg && (seg->ref >= 0 || seg->globalmv)) {
  ------------------
  |  Branch (1033:13): [True: 2.25M, False: 2.09M]
  |  Branch (1033:21): [True: 263k, False: 1.99M]
  |  Branch (1033:38): [True: 1.70M, False: 282k]
  ------------------
 1034|  1.97M|            b->intra = !seg->ref;
 1035|  2.37M|        } else {
 1036|  2.37M|            const int ictx = get_intra_ctx(t->a, &t->l, by4, bx4,
 1037|  2.37M|                                           have_top, have_left);
 1038|  2.37M|            b->intra = !dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  2.37M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1039|  2.37M|                            ts->cdf.m.intra[ictx]);
 1040|  2.37M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  2.37M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 2.37M]
  |  |  ------------------
  |  |   35|  2.37M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  2.37M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1041|      0|                printf("Post-intra[%d]: r=%d\n", b->intra, ts->msac.rng);
 1042|  2.37M|        }
 1043|  7.43M|    } else if (f->frame_hdr->allow_intrabc) {
  ------------------
  |  Branch (1043:16): [True: 5.91M, False: 1.52M]
  ------------------
 1044|  5.91M|        b->intra = !dav1d_msac_decode_bool_adapt(&ts->msac, ts->cdf.m.intrabc);
  ------------------
  |  |   52|  5.91M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1045|  5.91M|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  5.91M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 5.91M]
  |  |  ------------------
  |  |   35|  5.91M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  5.91M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1046|      0|            printf("Post-intrabcflag[%d]: r=%d\n", b->intra, ts->msac.rng);
 1047|  5.91M|    } else {
 1048|  1.52M|        b->intra = 1;
 1049|  1.52M|    }
 1050|       |
 1051|       |    // intra/inter-specific stuff
 1052|  11.8M|    if (b->intra) {
  ------------------
  |  Branch (1052:9): [True: 6.55M, False: 5.25M]
  ------------------
 1053|  6.55M|        uint16_t *const ymode_cdf = IS_INTER_OR_SWITCH(f->frame_hdr) ?
  ------------------
  |  |   36|  6.55M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 442k, False: 6.11M]
  |  |  ------------------
  ------------------
 1054|   442k|            ts->cdf.m.y_mode[dav1d_ymode_size_context[bs]] :
 1055|  6.55M|            ts->cdf.kfym[dav1d_intra_mode_context[t->a->mode[bx4]]]
 1056|  6.11M|                        [dav1d_intra_mode_context[t->l.mode[by4]]];
 1057|  6.55M|        b->y_mode = dav1d_msac_decode_symbol_adapt16(&ts->msac, ymode_cdf,
  ------------------
  |  |   57|  6.55M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1058|  6.55M|                                                     N_INTRA_PRED_MODES - 1);
 1059|  6.55M|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  6.55M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 6.55M]
  |  |  ------------------
  |  |   35|  6.55M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  6.55M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1060|      0|            printf("Post-ymode[%d]: r=%d\n", b->y_mode, ts->msac.rng);
 1061|       |
 1062|       |        // angle delta
 1063|  6.55M|        if (b_dim[2] + b_dim[3] >= 2 && b->y_mode >= VERT_PRED &&
  ------------------
  |  Branch (1063:13): [True: 5.40M, False: 1.14M]
  |  Branch (1063:41): [True: 2.75M, False: 2.65M]
  ------------------
 1064|  2.75M|            b->y_mode <= VERT_LEFT_PRED)
  ------------------
  |  Branch (1064:13): [True: 1.50M, False: 1.24M]
  ------------------
 1065|  1.50M|        {
 1066|  1.50M|            uint16_t *const acdf = ts->cdf.m.angle_delta[b->y_mode - VERT_PRED];
 1067|  1.50M|            const int angle = dav1d_msac_decode_symbol_adapt8(&ts->msac, acdf, 6);
  ------------------
  |  |   48|  1.50M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1068|  1.50M|            b->y_angle = angle - 3;
 1069|  5.04M|        } else {
 1070|  5.04M|            b->y_angle = 0;
 1071|  5.04M|        }
 1072|       |
 1073|  6.55M|        if (has_chroma) {
  ------------------
  |  Branch (1073:13): [True: 5.31M, False: 1.24M]
  ------------------
 1074|  5.31M|            const int cfl_allowed = f->frame_hdr->segmentation.lossless[b->seg_id] ?
  ------------------
  |  Branch (1074:37): [True: 127k, False: 5.18M]
  ------------------
 1075|  5.18M|                cbw4 == 1 && cbh4 == 1 : !!(cfl_allowed_mask & (1 << bs));
  ------------------
  |  Branch (1075:17): [True: 77.9k, False: 49.1k]
  |  Branch (1075:30): [True: 74.0k, False: 3.95k]
  ------------------
 1076|  5.31M|            uint16_t *const uvmode_cdf = ts->cdf.m.uv_mode[cfl_allowed][b->y_mode];
 1077|  5.31M|            b->uv_mode = dav1d_msac_decode_symbol_adapt16(&ts->msac, uvmode_cdf,
  ------------------
  |  |   57|  5.31M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1078|  5.31M|                             N_UV_INTRA_PRED_MODES - 1 - !cfl_allowed);
 1079|  5.31M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  5.31M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 5.31M]
  |  |  ------------------
  |  |   35|  5.31M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  5.31M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1080|      0|                printf("Post-uvmode[%d]: r=%d\n", b->uv_mode, ts->msac.rng);
 1081|       |
 1082|  5.31M|            b->uv_angle = 0;
 1083|  5.31M|            if (b->uv_mode == CFL_PRED) {
  ------------------
  |  Branch (1083:17): [True: 1.09M, False: 4.22M]
  ------------------
 1084|  1.09M|#define SIGN(a) (!!(a) + ((a) > 0))
 1085|  1.09M|                const int sign = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  1.09M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1086|  1.09M|                                     ts->cdf.m.cfl_sign, 7) + 1;
 1087|  1.09M|                const int sign_u = sign * 0x56 >> 8, sign_v = sign - sign_u * 3;
 1088|  1.09M|                assert(sign_u == sign / 3);
  ------------------
  |  Branch (1088:17): [True: 1.09M, False: 18.4E]
  ------------------
 1089|  1.09M|                if (sign_u) {
  ------------------
  |  Branch (1089:21): [True: 1.04M, False: 49.3k]
  ------------------
 1090|  1.04M|                    const int ctx = (sign_u == 2) * 3 + sign_v;
 1091|  1.04M|                    b->cfl_alpha[0] = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|  1.04M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1092|  1.04M|                                          ts->cdf.m.cfl_alpha[ctx], 15) + 1;
 1093|  1.04M|                    if (sign_u == 1) b->cfl_alpha[0] = -b->cfl_alpha[0];
  ------------------
  |  Branch (1093:25): [True: 616k, False: 426k]
  ------------------
 1094|  1.04M|                } else {
 1095|  49.3k|                    b->cfl_alpha[0] = 0;
 1096|  49.3k|                }
 1097|  1.09M|                if (sign_v) {
  ------------------
  |  Branch (1097:21): [True: 769k, False: 322k]
  ------------------
 1098|   769k|                    const int ctx = (sign_v == 2) * 3 + sign_u;
 1099|   769k|                    b->cfl_alpha[1] = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|   769k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1100|   769k|                                          ts->cdf.m.cfl_alpha[ctx], 15) + 1;
 1101|   769k|                    if (sign_v == 1) b->cfl_alpha[1] = -b->cfl_alpha[1];
  ------------------
  |  Branch (1101:25): [True: 271k, False: 498k]
  ------------------
 1102|   769k|                } else {
 1103|   322k|                    b->cfl_alpha[1] = 0;
 1104|   322k|                }
 1105|  1.09M|#undef SIGN
 1106|  1.09M|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.09M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.09M]
  |  |  ------------------
  |  |   35|  1.09M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.09M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1107|      0|                    printf("Post-uvalphas[%d/%d]: r=%d\n",
 1108|      0|                           b->cfl_alpha[0], b->cfl_alpha[1], ts->msac.rng);
 1109|  4.22M|            } else if (b_dim[2] + b_dim[3] >= 2 && b->uv_mode >= VERT_PRED &&
  ------------------
  |  Branch (1109:24): [True: 3.69M, False: 526k]
  |  Branch (1109:52): [True: 2.11M, False: 1.57M]
  ------------------
 1110|  2.11M|                       b->uv_mode <= VERT_LEFT_PRED)
  ------------------
  |  Branch (1110:24): [True: 1.09M, False: 1.01M]
  ------------------
 1111|  1.09M|            {
 1112|  1.09M|                uint16_t *const acdf = ts->cdf.m.angle_delta[b->uv_mode - VERT_PRED];
 1113|  1.09M|                const int angle = dav1d_msac_decode_symbol_adapt8(&ts->msac, acdf, 6);
  ------------------
  |  |   48|  1.09M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1114|  1.09M|                b->uv_angle = angle - 3;
 1115|  1.09M|            }
 1116|  5.31M|        }
 1117|       |
 1118|  6.55M|        b->pal_sz[0] = b->pal_sz[1] = 0;
 1119|  6.55M|        if (f->frame_hdr->allow_screen_content_tools &&
  ------------------
  |  Branch (1119:13): [True: 4.82M, False: 1.73M]
  ------------------
 1120|  4.82M|            imax(bw4, bh4) <= 16 && bw4 + bh4 >= 4)
  ------------------
  |  Branch (1120:13): [True: 4.48M, False: 339k]
  |  Branch (1120:37): [True: 3.77M, False: 710k]
  ------------------
 1121|  3.77M|        {
 1122|  3.77M|            const int sz_ctx = b_dim[2] + b_dim[3] - 2;
 1123|  3.77M|            if (b->y_mode == DC_PRED) {
  ------------------
  |  Branch (1123:17): [True: 1.88M, False: 1.88M]
  ------------------
 1124|  1.88M|                const int pal_ctx = (t->a->pal_sz[bx4] > 0) + (t->l.pal_sz[by4] > 0);
 1125|  1.88M|                const int use_y_pal = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  1.88M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1126|  1.88M|                                          ts->cdf.m.pal_y[sz_ctx][pal_ctx]);
 1127|  1.88M|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.88M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.88M]
  |  |  ------------------
  |  |   35|  1.88M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.88M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1128|      0|                    printf("Post-y_pal[%d]: r=%d\n", use_y_pal, ts->msac.rng);
 1129|  1.88M|                if (use_y_pal)
  ------------------
  |  Branch (1129:21): [True: 165k, False: 1.71M]
  ------------------
 1130|   165k|                    f->bd_fn.read_pal_plane(t, b, 0, sz_ctx, bx4, by4);
 1131|  1.88M|            }
 1132|       |
 1133|  3.77M|            if (has_chroma && b->uv_mode == DC_PRED) {
  ------------------
  |  Branch (1133:17): [True: 3.33M, False: 436k]
  |  Branch (1133:31): [True: 1.08M, False: 2.25M]
  ------------------
 1134|  1.08M|                const int pal_ctx = b->pal_sz[0] > 0;
 1135|  1.08M|                const int use_uv_pal = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  1.08M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1136|  1.08M|                                           ts->cdf.m.pal_uv[pal_ctx]);
 1137|  1.08M|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.08M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.08M]
  |  |  ------------------
  |  |   35|  1.08M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.08M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1138|      0|                    printf("Post-uv_pal[%d]: r=%d\n", use_uv_pal, ts->msac.rng);
 1139|  1.08M|                if (use_uv_pal) // see aomedia bug 2183 for why we use luma coordinates
  ------------------
  |  Branch (1139:21): [True: 33.9k, False: 1.04M]
  ------------------
 1140|  33.9k|                    f->bd_fn.read_pal_uv(t, b, sz_ctx, bx4, by4);
 1141|  1.08M|            }
 1142|  3.77M|        }
 1143|       |
 1144|  6.55M|        if (b->y_mode == DC_PRED && !b->pal_sz[0] &&
  ------------------
  |  Branch (1144:13): [True: 3.12M, False: 3.42M]
  |  Branch (1144:37): [True: 2.96M, False: 165k]
  ------------------
 1145|  2.96M|            imax(b_dim[2], b_dim[3]) <= 3 && f->seq_hdr->filter_intra)
  ------------------
  |  Branch (1145:13): [True: 2.07M, False: 886k]
  |  Branch (1145:46): [True: 1.40M, False: 668k]
  ------------------
 1146|  1.40M|        {
 1147|  1.40M|            const int is_filter = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  1.40M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1148|  1.40M|                                      ts->cdf.m.use_filter_intra[bs]);
 1149|  1.40M|            if (is_filter) {
  ------------------
  |  Branch (1149:17): [True: 913k, False: 495k]
  ------------------
 1150|   913k|                b->y_mode = FILTER_PRED;
 1151|   913k|                b->y_angle = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   913k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1152|   913k|                                 ts->cdf.m.filter_intra, 4);
 1153|   913k|            }
 1154|  1.40M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.40M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.40M]
  |  |  ------------------
  |  |   35|  1.40M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.40M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1155|      0|                printf("Post-filterintramode[%d/%d]: r=%d\n",
 1156|      0|                       b->y_mode, b->y_angle, ts->msac.rng);
 1157|  1.40M|        }
 1158|       |
 1159|  6.55M|        if (b->pal_sz[0]) {
  ------------------
  |  Branch (1159:13): [True: 166k, False: 6.39M]
  ------------------
 1160|   166k|            uint8_t *pal_idx;
 1161|   166k|            if (t->frame_thread.pass) {
  ------------------
  |  Branch (1161:17): [True: 165k, False: 6]
  ------------------
 1162|   165k|                const int p = t->frame_thread.pass & 1;
 1163|   165k|                assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1163:17): [True: 166k, False: 18.4E]
  ------------------
 1164|   166k|                pal_idx = ts->frame_thread[p].pal_idx;
 1165|   166k|                ts->frame_thread[p].pal_idx += bw4 * bh4 * 8;
 1166|   166k|            } else
 1167|      6|                pal_idx = t->scratch.pal_idx_y;
 1168|   166k|            read_pal_indices(t, pal_idx, b->pal_sz[0], 0, w4, h4, bw4, bh4);
 1169|   166k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   166k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 166k]
  |  |  ------------------
  |  |   35|   166k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   166k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1170|      0|                printf("Post-y-pal-indices: r=%d\n", ts->msac.rng);
 1171|   166k|        }
 1172|       |
 1173|  6.55M|        if (has_chroma && b->pal_sz[1]) {
  ------------------
  |  Branch (1173:13): [True: 5.32M, False: 1.23M]
  |  Branch (1173:27): [True: 33.9k, False: 5.28M]
  ------------------
 1174|  33.9k|            uint8_t *pal_idx;
 1175|  33.9k|            if (t->frame_thread.pass) {
  ------------------
  |  Branch (1175:17): [True: 33.9k, False: 0]
  ------------------
 1176|  33.9k|                const int p = t->frame_thread.pass & 1;
 1177|  33.9k|                assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1177:17): [True: 33.9k, False: 18.4E]
  ------------------
 1178|  33.9k|                pal_idx = ts->frame_thread[p].pal_idx;
 1179|  33.9k|                ts->frame_thread[p].pal_idx += cbw4 * cbh4 * 8;
 1180|  33.9k|            } else
 1181|      0|                pal_idx = t->scratch.pal_idx_uv;
 1182|  33.9k|            read_pal_indices(t, pal_idx, b->pal_sz[1], 1, cw4, ch4, cbw4, cbh4);
 1183|  33.9k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  33.9k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 33.9k]
  |  |  ------------------
  |  |   35|  33.9k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  33.9k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1184|      0|                printf("Post-uv-pal-indices: r=%d\n", ts->msac.rng);
 1185|  33.9k|        }
 1186|       |
 1187|  6.55M|        const TxfmInfo *t_dim;
 1188|  6.55M|        if (f->frame_hdr->segmentation.lossless[b->seg_id]) {
  ------------------
  |  Branch (1188:13): [True: 197k, False: 6.35M]
  ------------------
 1189|   197k|            b->tx = b->uvtx = (int) TX_4X4;
 1190|   197k|            t_dim = &dav1d_txfm_dimensions[TX_4X4];
 1191|  6.35M|        } else {
 1192|  6.35M|            b->tx = dav1d_max_txfm_size_for_bs[bs][0];
 1193|  6.35M|            b->uvtx = dav1d_max_txfm_size_for_bs[bs][f->cur.p.layout];
 1194|  6.35M|            t_dim = &dav1d_txfm_dimensions[b->tx];
 1195|  6.35M|            if (f->frame_hdr->txfm_mode == DAV1D_TX_SWITCHABLE && t_dim->max > TX_4X4) {
  ------------------
  |  Branch (1195:17): [True: 1.03M, False: 5.32M]
  |  Branch (1195:67): [True: 933k, False: 97.2k]
  ------------------
 1196|   933k|                const int tctx = get_tx_ctx(t->a, &t->l, t_dim, by4, bx4);
 1197|   933k|                uint16_t *const tx_cdf = ts->cdf.m.txsz[t_dim->max - 1][tctx];
 1198|   933k|                int depth = dav1d_msac_decode_symbol_adapt4(&ts->msac, tx_cdf,
  ------------------
  |  |   47|   933k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1199|   933k|                                imin(t_dim->max, 2));
 1200|       |
 1201|  1.36M|                while (depth--) {
  ------------------
  |  Branch (1201:24): [True: 432k, False: 933k]
  ------------------
 1202|   432k|                    b->tx = t_dim->sub;
 1203|   432k|                    t_dim = &dav1d_txfm_dimensions[b->tx];
 1204|   432k|                }
 1205|   933k|            }
 1206|  6.35M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  6.35M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 6.35M]
  |  |  ------------------
  |  |   35|  6.35M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  6.35M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1207|      0|                printf("Post-tx[%d]: r=%d\n", b->tx, ts->msac.rng);
 1208|  6.35M|        }
 1209|       |
 1210|       |        // reconstruction
 1211|  6.56M|        if (t->frame_thread.pass == 1) {
  ------------------
  |  Branch (1211:13): [True: 6.56M, False: 18.4E]
  ------------------
 1212|  6.56M|            f->bd_fn.read_coef_blocks(t, bs, b);
 1213|  18.4E|        } else {
 1214|  18.4E|            f->bd_fn.recon_b_intra(t, bs, intra_edge_flags, b);
 1215|  18.4E|        }
 1216|       |
 1217|  6.55M|        if (f->frame_hdr->loopfilter.level_y[0] ||
  ------------------
  |  Branch (1217:13): [True: 1.30M, False: 5.25M]
  ------------------
 1218|  5.25M|            f->frame_hdr->loopfilter.level_y[1])
  ------------------
  |  Branch (1218:13): [True: 218k, False: 5.03M]
  ------------------
 1219|  1.53M|        {
 1220|  1.53M|            dav1d_create_lf_mask_intra(t->lf_mask, f->lf.level, f->b4_stride,
 1221|  1.53M|                                       (const uint8_t (*)[8][2])
 1222|  1.53M|                                       &ts->lflvl[b->seg_id][0][0][0],
 1223|  1.53M|                                       t->bx, t->by, f->w4, f->h4, bs,
 1224|  1.53M|                                       b->tx, b->uvtx, f->cur.p.layout,
 1225|  1.53M|                                       &t->a->tx_lpf_y[bx4], &t->l.tx_lpf_y[by4],
 1226|  1.53M|                                       has_chroma ? &t->a->tx_lpf_uv[cbx4] : NULL,
  ------------------
  |  Branch (1226:40): [True: 1.08M, False: 452k]
  ------------------
 1227|  1.53M|                                       has_chroma ? &t->l.tx_lpf_uv[cby4] : NULL);
  ------------------
  |  Branch (1227:40): [True: 1.08M, False: 452k]
  ------------------
 1228|  1.53M|        }
 1229|       |        // update contexts
 1230|  6.55M|        const enum IntraPredMode y_mode_nofilt =
 1231|  6.55M|            b->y_mode == FILTER_PRED ? DC_PRED : b->y_mode;
  ------------------
  |  Branch (1231:13): [True: 913k, False: 5.64M]
  ------------------
 1232|  6.55M|        BlockContext *edge = t->a;
 1233|  19.6M|        for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (1233:36): [True: 13.1M, False: 6.55M]
  ------------------
 1234|  13.1M|            int t_lsz = ((uint8_t *) &t_dim->lw)[i]; // lw then lh
 1235|  13.1M|#define set_ctx(rep_macro) \
 1236|  13.1M|            rep_macro(edge->tx_intra, off, t_lsz); \
 1237|  13.1M|            rep_macro(edge->tx, off, t_lsz); \
 1238|  13.1M|            rep_macro(edge->mode, off, y_mode_nofilt); \
 1239|  13.1M|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
 1240|  13.1M|            rep_macro(edge->seg_pred, off, seg_pred); \
 1241|  13.1M|            rep_macro(edge->skip_mode, off, 0); \
 1242|  13.1M|            rep_macro(edge->intra, off, 1); \
 1243|  13.1M|            rep_macro(edge->skip, off, b->skip); \
 1244|       |            /* see aomedia bug 2183 for why we use luma coordinates here */ \
 1245|  13.1M|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
 1246|  13.1M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
 1247|  13.1M|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
 1248|  13.1M|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
 1249|  13.1M|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
 1250|  13.1M|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
 1251|  13.1M|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
 1252|  13.1M|            }
 1253|  13.1M|            case_set(b_dim[2 + i]);
  ------------------
  |  |   70|  13.1M|    switch (var) { \
  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 1236|  2.31M|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  2.31M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|  2.31M|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  2.31M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|  2.31M|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  2.31M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|  2.31M|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  2.31M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|  2.31M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  2.31M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|  2.31M|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  2.31M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|  2.31M|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  2.31M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|  2.31M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  2.31M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|  2.31M|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|  2.31M|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|  2.31M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  4.63M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (56:43): [True: 1.54M, False: 772k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|  2.31M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  2.31M|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 198k, False: 2.11M]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|   198k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   198k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   198k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|   198k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   198k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   198k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|   198k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   198k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   198k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|   198k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   198k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   198k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|   198k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   198k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   198k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|   198k|            }
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 2.31M, False: 10.8M]
  |  |  ------------------
  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 1236|  3.81M|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  3.81M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|  3.81M|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  3.81M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|  3.81M|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  3.81M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|  3.81M|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  3.81M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|  3.81M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  3.81M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|  3.81M|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  3.81M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|  3.81M|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  3.81M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|  3.81M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  3.81M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|  3.81M|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|  3.81M|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  3.81M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  7.63M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (58:45): [True: 3.39M, False: 426k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|  3.81M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  3.81M|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 312k, False: 3.50M]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|   312k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   312k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   312k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|   312k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   312k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   312k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|   312k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   312k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   312k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|   312k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   312k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   312k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|   312k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   312k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   312k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|   312k|            }
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 3.81M, False: 9.31M]
  |  |  ------------------
  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 1236|  3.20M|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  3.20M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|  3.20M|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  3.20M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|  3.20M|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  3.20M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|  3.20M|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  3.20M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|  3.20M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  3.20M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|  3.20M|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  3.20M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|  3.20M|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  3.20M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|  3.20M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  3.20M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|  3.20M|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|  3.20M|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  3.20M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  6.40M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (60:45): [True: 2.84M, False: 355k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|  3.20M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  3.20M|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 231k, False: 2.97M]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|   231k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   231k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   231k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|   231k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   231k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   231k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|   231k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   231k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   231k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|   231k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   231k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   231k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|   231k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   231k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   231k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|   231k|            }
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 3.20M, False: 9.92M]
  |  |  ------------------
  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 1236|  1.78M|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.78M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|  1.78M|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.78M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|  1.78M|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.78M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|  1.78M|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.78M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|  1.78M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.78M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|  1.78M|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.78M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|  1.78M|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.78M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|  1.78M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.78M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|  1.78M|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|  1.78M|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  1.78M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  3.57M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (62:45): [True: 1.65M, False: 134k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|  1.78M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  1.78M|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 73.3k, False: 1.71M]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|  73.3k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  73.3k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  73.3k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|  73.3k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  73.3k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  73.3k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|  73.3k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  73.3k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  73.3k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|  73.3k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  73.3k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  73.3k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|  73.3k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|  73.3k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  73.3k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|  73.3k|            }
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 1.78M, False: 11.3M]
  |  |  ------------------
  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 1236|  1.29M|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.29M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.29M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.29M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.29M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|  1.29M|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.29M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.29M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.29M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.29M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|  1.29M|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.29M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.29M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.29M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.29M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|  1.29M|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.29M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.29M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.29M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.29M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|  1.29M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.29M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.29M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.29M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.29M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|  1.29M|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.29M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.29M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.29M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.29M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|  1.29M|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.29M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.29M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.29M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.29M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|  1.29M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.29M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  1.29M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  1.29M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.29M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|  1.29M|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|  1.29M|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  1.29M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  1.29M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.58M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (64:29): [True: 684k, False: 607k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   65|  1.29M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 1.29M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|  1.29M|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  1.29M|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 38.8k, False: 1.25M]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|  38.8k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  38.8k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  38.8k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  38.8k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  38.8k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 38.8k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|  38.8k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  38.8k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  38.8k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  38.8k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  38.8k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 38.8k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|  38.8k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  38.8k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  38.8k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  38.8k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  38.8k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 38.8k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|  38.8k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  38.8k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  38.8k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  38.8k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  38.8k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 38.8k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|  38.8k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  38.8k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  38.8k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  38.8k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  38.8k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 38.8k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|  38.8k|            }
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 1.29M, False: 11.8M]
  |  |  ------------------
  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  | 1236|   746k|            rep_macro(edge->tx_intra, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   746k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   746k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   746k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 746k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1237|   746k|            rep_macro(edge->tx, off, t_lsz); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   746k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   746k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   746k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 746k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1238|   746k|            rep_macro(edge->mode, off, y_mode_nofilt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   746k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   746k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   746k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 746k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1239|   746k|            rep_macro(edge->pal_sz, off, b->pal_sz[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   746k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   746k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   746k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 746k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1240|   746k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   746k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   746k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   746k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 746k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1241|   746k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   746k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   746k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   746k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 746k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1242|   746k|            rep_macro(edge->intra, off, 1); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   746k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   746k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   746k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 746k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1243|   746k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   746k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|   746k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|   746k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 746k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1244|   746k|            /* see aomedia bug 2183 for why we use luma coordinates here */ \
  |  |  |  | 1245|   746k|            rep_macro(t->pal_sz_uv[i], off, (has_chroma ? b->pal_sz[1] : 0)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|   746k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|   746k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.49M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (67:29): [True: 544k, False: 202k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   68|   746k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 746k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1246|   746k|            if (IS_INTER_OR_SWITCH(f->frame_hdr)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|   746k|    ((frame_header)->frame_type & 1)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  Branch (36:5): [True: 30.4k, False: 716k]
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1247|  30.4k|                rep_macro(edge->comp_type, off, COMP_INTER_NONE); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  30.4k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  30.4k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  30.4k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  30.4k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 30.4k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1248|  30.4k|                rep_macro(edge->ref[0], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  30.4k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  30.4k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  30.4k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  30.4k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 30.4k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1249|  30.4k|                rep_macro(edge->ref[1], off, ((uint8_t) -1)); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  30.4k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  30.4k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  30.4k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  30.4k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 30.4k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1250|  30.4k|                rep_macro(edge->filter[0], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  30.4k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  30.4k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  30.4k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  30.4k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 30.4k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1251|  30.4k|                rep_macro(edge->filter[1], off, DAV1D_N_SWITCHABLE_FILTERS); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  30.4k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  30.4k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  30.4k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  30.4k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 30.4k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1252|  30.4k|            }
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 746k, False: 12.3M]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 13.1M]
  |  |  ------------------
  |  |   78|  13.1M|    }
  ------------------
  |  Branch (1253:13): [Folded, False: 0]
  ------------------
 1254|  13.1M|#undef set_ctx
 1255|  13.1M|        }
 1256|  6.55M|        if (b->pal_sz[0])
  ------------------
  |  Branch (1256:13): [True: 165k, False: 6.38M]
  ------------------
 1257|   165k|            f->bd_fn.copy_pal_block_y(t, bx4, by4, bw4, bh4);
 1258|  6.55M|        if (has_chroma) {
  ------------------
  |  Branch (1258:13): [True: 5.32M, False: 1.23M]
  ------------------
 1259|  5.32M|            uint8_t uv_mode = b->uv_mode;
 1260|  5.32M|            dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], uv_mode);
 1261|  5.32M|            dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], uv_mode);
 1262|  5.32M|            if (b->pal_sz[1])
  ------------------
  |  Branch (1262:17): [True: 33.9k, False: 5.28M]
  ------------------
 1263|  33.9k|                f->bd_fn.copy_pal_block_uv(t, bx4, by4, bw4, bh4);
 1264|  5.32M|        }
 1265|  6.55M|        if (IS_INTER_OR_SWITCH(f->frame_hdr) || f->frame_hdr->allow_intrabc)
  ------------------
  |  |   36|  13.1M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 419k, False: 6.13M]
  |  |  ------------------
  ------------------
  |  Branch (1265:49): [True: 4.61M, False: 1.52M]
  ------------------
 1266|  5.05M|            splat_intraref(f->c, t, bs, bw4, bh4);
 1267|  6.55M|    } else if (IS_KEY_OR_INTRA(f->frame_hdr)) {
  ------------------
  |  |   43|  5.25M|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  5.25M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 1.33M, False: 3.92M]
  |  |  ------------------
  ------------------
 1268|       |        // intra block copy
 1269|  1.33M|        refmvs_candidate mvstack[8];
 1270|  1.33M|        int n_mvs, ctx;
 1271|  1.33M|        dav1d_refmvs_find(&t->rt, mvstack, &n_mvs, &ctx,
 1272|  1.33M|                          (union refmvs_refpair) { .ref = { 0, -1 }},
 1273|  1.33M|                          bs, intra_edge_flags, t->by, t->bx);
 1274|       |
 1275|  1.33M|        if (mvstack[0].mv.mv[0].n)
  ------------------
  |  Branch (1275:13): [True: 1.09M, False: 238k]
  ------------------
 1276|  1.09M|            b->mv[0] = mvstack[0].mv.mv[0];
 1277|   238k|        else if (mvstack[1].mv.mv[0].n)
  ------------------
  |  Branch (1277:18): [True: 0, False: 238k]
  ------------------
 1278|      0|            b->mv[0] = mvstack[1].mv.mv[0];
 1279|   238k|        else {
 1280|   238k|            if (t->by - (16 << f->seq_hdr->sb128) < ts->tiling.row_start) {
  ------------------
  |  Branch (1280:17): [True: 236k, False: 1.30k]
  ------------------
 1281|   236k|                b->mv[0].y = 0;
 1282|   236k|                b->mv[0].x = -(512 << f->seq_hdr->sb128) - 2048;
 1283|   236k|            } else {
 1284|  1.30k|                b->mv[0].y = -(512 << f->seq_hdr->sb128);
 1285|  1.30k|                b->mv[0].x = 0;
 1286|  1.30k|            }
 1287|   238k|        }
 1288|       |
 1289|  1.33M|        const union mv ref = b->mv[0];
 1290|  1.33M|        read_mv_residual(ts, &b->mv[0], -1);
 1291|       |
 1292|       |        // clip intrabc motion vector to decoded parts of current tile
 1293|  1.33M|        int border_left = ts->tiling.col_start * 4;
 1294|  1.33M|        int border_top  = ts->tiling.row_start * 4;
 1295|  1.33M|        if (has_chroma) {
  ------------------
  |  Branch (1295:13): [True: 1.24M, False: 89.4k]
  ------------------
 1296|  1.24M|            if (bw4 < 2 &&  ss_hor)
  ------------------
  |  Branch (1296:17): [True: 284k, False: 961k]
  |  Branch (1296:29): [True: 22.2k, False: 262k]
  ------------------
 1297|  22.2k|                border_left += 4;
 1298|  1.24M|            if (bh4 < 2 &&  ss_ver)
  ------------------
  |  Branch (1298:17): [True: 298k, False: 948k]
  |  Branch (1298:29): [True: 35.4k, False: 262k]
  ------------------
 1299|  35.4k|                border_top  += 4;
 1300|  1.24M|        }
 1301|  1.33M|        int src_left   = t->bx * 4 + (b->mv[0].x >> 3);
 1302|  1.33M|        int src_top    = t->by * 4 + (b->mv[0].y >> 3);
 1303|  1.33M|        int src_right  = src_left + bw4 * 4;
 1304|  1.33M|        int src_bottom = src_top  + bh4 * 4;
 1305|  1.33M|        const int border_right = ((ts->tiling.col_end + (bw4 - 1)) & ~(bw4 - 1)) * 4;
 1306|       |
 1307|       |        // check against left or right tile boundary and adjust if necessary
 1308|  1.33M|        if (src_left < border_left) {
  ------------------
  |  Branch (1308:13): [True: 158k, False: 1.17M]
  ------------------
 1309|   158k|            src_right += border_left - src_left;
 1310|   158k|            src_left  += border_left - src_left;
 1311|  1.17M|        } else if (src_right > border_right) {
  ------------------
  |  Branch (1311:20): [True: 289k, False: 888k]
  ------------------
 1312|   289k|            src_left  -= src_right - border_right;
 1313|   289k|            src_right -= src_right - border_right;
 1314|   289k|        }
 1315|       |        // check against top tile boundary and adjust if necessary
 1316|  1.33M|        if (src_top < border_top) {
  ------------------
  |  Branch (1316:13): [True: 682k, False: 653k]
  ------------------
 1317|   682k|            src_bottom += border_top - src_top;
 1318|   682k|            src_top    += border_top - src_top;
 1319|   682k|        }
 1320|       |
 1321|  1.33M|        const int sbx = (t->bx >> (4 + f->seq_hdr->sb128)) << (6 + f->seq_hdr->sb128);
 1322|  1.33M|        const int sby = (t->by >> (4 + f->seq_hdr->sb128)) << (6 + f->seq_hdr->sb128);
 1323|  1.33M|        const int sb_size = 1 << (6 + f->seq_hdr->sb128);
 1324|       |        // check for overlap with current superblock
 1325|  1.33M|        if (src_bottom > sby && src_right > sbx) {
  ------------------
  |  Branch (1325:13): [True: 1.31M, False: 23.9k]
  |  Branch (1325:33): [True: 302k, False: 1.01M]
  ------------------
 1326|   302k|            if (src_top - border_top >= src_bottom - sby) {
  ------------------
  |  Branch (1326:17): [True: 1.56k, False: 300k]
  ------------------
 1327|       |                // if possible move src up into the previous suberblock row
 1328|  1.56k|                src_top    -= src_bottom - sby;
 1329|  1.56k|                src_bottom -= src_bottom - sby;
 1330|   300k|            } else if (src_left - border_left >= src_right - sbx) {
  ------------------
  |  Branch (1330:24): [True: 288k, False: 12.3k]
  ------------------
 1331|       |                // if possible move src left into the previous suberblock
 1332|   288k|                src_left  -= src_right - sbx;
 1333|   288k|                src_right -= src_right - sbx;
 1334|   288k|            }
 1335|   302k|        }
 1336|       |        // move src up if it is below current superblock row
 1337|  1.33M|        if (src_bottom > sby + sb_size) {
  ------------------
  |  Branch (1337:13): [True: 19.4k, False: 1.31M]
  ------------------
 1338|  19.4k|            src_top    -= src_bottom - (sby + sb_size);
 1339|  19.4k|            src_bottom -= src_bottom - (sby + sb_size);
 1340|  19.4k|        }
 1341|       |        // error out if mv still overlaps with the current superblock
 1342|  1.33M|        if (src_bottom > sby && src_right > sbx)
  ------------------
  |  Branch (1342:13): [True: 1.31M, False: 25.5k]
  |  Branch (1342:33): [True: 12.3k, False: 1.29M]
  ------------------
 1343|  12.3k|            return -1;
 1344|       |
 1345|  1.32M|        b->mv[0].x = (src_left - t->bx * 4) * 8;
 1346|  1.32M|        b->mv[0].y = (src_top  - t->by * 4) * 8;
 1347|       |
 1348|  1.32M|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.32M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.32M]
  |  |  ------------------
  |  |   35|  1.32M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.32M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1349|      0|            printf("Post-dmv[%d/%d,ref=%d/%d|%d/%d]: r=%d\n",
 1350|      0|                   b->mv[0].y, b->mv[0].x, ref.y, ref.x,
 1351|      0|                   mvstack[0].mv.mv[0].y, mvstack[0].mv.mv[0].x, ts->msac.rng);
 1352|  1.32M|        read_vartx_tree(t, b, bs, bx4, by4);
 1353|       |
 1354|       |        // reconstruction
 1355|  1.32M|        if (t->frame_thread.pass == 1) {
  ------------------
  |  Branch (1355:13): [True: 1.32M, False: 18.4E]
  ------------------
 1356|  1.32M|            f->bd_fn.read_coef_blocks(t, bs, b);
 1357|  1.32M|            b->filter2d = FILTER_2D_BILINEAR;
 1358|  18.4E|        } else {
 1359|  18.4E|            if (f->bd_fn.recon_b_inter(t, bs, b)) return -1;
  ------------------
  |  Branch (1359:17): [True: 0, False: 18.4E]
  ------------------
 1360|  18.4E|        }
 1361|       |
 1362|  1.32M|        splat_intrabc_mv(f->c, t, bs, b, bw4, bh4);
 1363|  1.32M|        BlockContext *edge = t->a;
 1364|  3.97M|        for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (1364:36): [True: 2.64M, False: 1.32M]
  ------------------
 1365|  2.64M|#define set_ctx(rep_macro) \
 1366|  2.64M|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
 1367|  2.64M|            rep_macro(edge->mode, off, DC_PRED); \
 1368|  2.64M|            rep_macro(edge->pal_sz, off, 0); \
 1369|       |            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
 1370|  2.64M|            rep_macro(t->pal_sz_uv[i], off, 0); \
 1371|  2.64M|            rep_macro(edge->seg_pred, off, seg_pred); \
 1372|  2.64M|            rep_macro(edge->skip_mode, off, 0); \
 1373|  2.64M|            rep_macro(edge->intra, off, 0); \
 1374|  2.64M|            rep_macro(edge->skip, off, b->skip)
 1375|  2.64M|            case_set(b_dim[2 + i]);
  ------------------
  |  |   70|  2.64M|    switch (var) { \
  |  |   71|   689k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 1366|   689k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   689k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   689k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   689k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   689k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   689k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   689k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   689k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   689k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   689k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   689k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   689k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   689k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   689k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   689k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   689k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   689k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   689k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   689k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   689k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   689k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   689k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   689k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   689k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   689k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 689k, False: 1.95M]
  |  |  ------------------
  |  |   72|   562k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 1366|   562k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   562k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   562k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   562k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   562k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   562k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   562k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   562k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   562k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   562k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   562k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   562k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   562k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   562k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   562k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   562k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   562k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   562k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   562k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   562k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   562k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   562k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   562k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   562k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   562k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 562k, False: 2.08M]
  |  |  ------------------
  |  |   73|   517k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 1366|   517k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   517k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   517k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   517k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   517k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   517k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   517k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   517k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   517k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   517k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   517k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   517k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   517k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   517k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   517k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   517k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   517k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   517k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   517k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   517k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   517k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   517k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   517k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|   517k|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|   517k|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 517k, False: 2.12M]
  |  |  ------------------
  |  |   74|   225k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 1366|   225k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   225k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   225k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   225k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   225k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   225k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   225k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   225k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   225k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   225k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   225k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   225k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   225k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   225k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   225k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   225k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   225k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   225k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   225k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   225k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   225k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   225k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   225k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   225k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   225k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 225k, False: 2.42M]
  |  |  ------------------
  |  |   75|   555k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 1366|   555k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   555k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   555k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   555k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   555k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 555k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|   555k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   555k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   555k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   555k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   555k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 555k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|   555k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   555k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   555k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   555k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   555k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 555k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|   555k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|   555k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   555k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   555k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   555k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   555k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 555k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|   555k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   555k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   555k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   555k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   555k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 555k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|   555k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   555k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   555k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   555k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   555k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 555k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|   555k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   555k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   555k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   555k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   555k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 555k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|   555k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|   555k|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|   555k|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|   555k|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|   555k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 555k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 555k, False: 2.09M]
  |  |  ------------------
  |  |   76|  99.6k|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  | 1366|  99.6k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  99.6k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  99.6k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  99.6k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  99.6k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 99.6k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1367|  99.6k|            rep_macro(edge->mode, off, DC_PRED); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  99.6k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  99.6k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  99.6k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  99.6k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 99.6k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1368|  99.6k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  99.6k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  99.6k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  99.6k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  99.6k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 99.6k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1369|  99.6k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1370|  99.6k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  99.6k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  99.6k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  99.6k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  99.6k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 99.6k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1371|  99.6k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  99.6k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  99.6k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  99.6k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  99.6k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 99.6k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1372|  99.6k|            rep_macro(edge->skip_mode, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  99.6k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  99.6k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  99.6k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  99.6k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 99.6k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1373|  99.6k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  99.6k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  99.6k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  99.6k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  99.6k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 99.6k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1374|  99.6k|            rep_macro(edge->skip, off, b->skip)
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  99.6k|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  99.6k|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  99.6k|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  99.6k|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 99.6k]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 99.6k, False: 2.54M]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 2.64M]
  |  |  ------------------
  |  |   78|  2.64M|    }
  ------------------
  |  Branch (1375:13): [Folded, False: 0]
  ------------------
 1376|  2.64M|#undef set_ctx
 1377|  2.64M|        }
 1378|  1.32M|        if (has_chroma) {
  ------------------
  |  Branch (1378:13): [True: 1.23M, False: 86.5k]
  ------------------
 1379|  1.23M|            dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], DC_PRED);
 1380|  1.23M|            dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], DC_PRED);
 1381|  1.23M|        }
 1382|  3.92M|    } else {
 1383|       |        // inter-specific mode/mv coding
 1384|  3.92M|        int is_comp, has_subpel_filter;
 1385|       |
 1386|  3.92M|        if (b->skip_mode) {
  ------------------
  |  Branch (1386:13): [True: 25.3k, False: 3.89M]
  ------------------
 1387|  25.3k|            is_comp = 1;
 1388|  3.89M|        } else if ((!seg || (seg->ref == -1 && !seg->globalmv && !seg->skip)) &&
  ------------------
  |  Branch (1388:21): [True: 1.71M, False: 2.18M]
  |  Branch (1388:30): [True: 1.93M, False: 254k]
  |  Branch (1388:48): [True: 220k, False: 1.70M]
  |  Branch (1388:66): [True: 196k, False: 23.9k]
  ------------------
 1389|  1.92M|                   f->frame_hdr->switchable_comp_refs && imin(bw4, bh4) > 1)
  ------------------
  |  Branch (1389:20): [True: 1.05M, False: 870k]
  |  Branch (1389:58): [True: 786k, False: 265k]
  ------------------
 1390|   786k|        {
 1391|   786k|            const int ctx = get_comp_ctx(t->a, &t->l, by4, bx4,
 1392|   786k|                                         have_top, have_left);
 1393|   786k|            is_comp = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   786k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1394|   786k|                          ts->cdf.m.comp[ctx]);
 1395|   786k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   786k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 786k]
  |  |  ------------------
  |  |   35|   786k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   786k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1396|      0|                printf("Post-compflag[%d]: r=%d\n", is_comp, ts->msac.rng);
 1397|  3.10M|        } else {
 1398|  3.10M|            is_comp = 0;
 1399|  3.10M|        }
 1400|       |
 1401|  3.92M|        if (b->skip_mode) {
  ------------------
  |  Branch (1401:13): [True: 25.3k, False: 3.89M]
  ------------------
 1402|  25.3k|            b->ref[0] = f->frame_hdr->skip_mode_refs[0];
 1403|  25.3k|            b->ref[1] = f->frame_hdr->skip_mode_refs[1];
 1404|  25.3k|            b->comp_type = COMP_INTER_AVG;
 1405|  25.3k|            b->inter_mode = NEARESTMV_NEARESTMV;
 1406|  25.3k|            b->drl_idx = NEAREST_DRL;
 1407|  25.3k|            has_subpel_filter = 0;
 1408|       |
 1409|  25.3k|            refmvs_candidate mvstack[8];
 1410|  25.3k|            int n_mvs, ctx;
 1411|  25.3k|            dav1d_refmvs_find(&t->rt, mvstack, &n_mvs, &ctx,
 1412|  25.3k|                              (union refmvs_refpair) { .ref = {
 1413|  25.3k|                                    b->ref[0] + 1, b->ref[1] + 1 }},
 1414|  25.3k|                              bs, intra_edge_flags, t->by, t->bx);
 1415|       |
 1416|  25.3k|            b->mv[0] = mvstack[0].mv.mv[0];
 1417|  25.3k|            b->mv[1] = mvstack[0].mv.mv[1];
 1418|  25.3k|            fix_mv_precision(f->frame_hdr, &b->mv[0]);
 1419|  25.3k|            fix_mv_precision(f->frame_hdr, &b->mv[1]);
 1420|  25.3k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  25.3k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 25.3k]
  |  |  ------------------
  |  |   35|  25.3k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  25.3k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1421|      0|                printf("Post-skipmodeblock[mv=1:y=%d,x=%d,2:y=%d,x=%d,refs=%d+%d\n",
 1422|      0|                       b->mv[0].y, b->mv[0].x, b->mv[1].y, b->mv[1].x,
 1423|      0|                       b->ref[0], b->ref[1]);
 1424|  3.89M|        } else if (is_comp) {
  ------------------
  |  Branch (1424:20): [True: 372k, False: 3.52M]
  ------------------
 1425|   372k|            const int dir_ctx = get_comp_dir_ctx(t->a, &t->l, by4, bx4,
 1426|   372k|                                                 have_top, have_left);
 1427|   372k|            if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   372k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1427:17): [True: 302k, False: 69.7k]
  ------------------
 1428|   372k|                    ts->cdf.m.comp_dir[dir_ctx]))
 1429|   302k|            {
 1430|       |                // bidir - first reference (fw)
 1431|   302k|                const int ctx1 = av1_get_fwd_ref_ctx(t->a, &t->l, by4, bx4,
 1432|   302k|                                                     have_top, have_left);
 1433|   302k|                if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   302k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1433:21): [True: 99.1k, False: 203k]
  ------------------
 1434|   302k|                        ts->cdf.m.comp_fwd_ref[0][ctx1]))
 1435|  99.1k|                {
 1436|  99.1k|                    const int ctx2 = av1_get_fwd_ref_2_ctx(t->a, &t->l, by4, bx4,
 1437|  99.1k|                                                           have_top, have_left);
 1438|  99.1k|                    b->ref[0] = 2 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  99.1k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1439|  99.1k|                                        ts->cdf.m.comp_fwd_ref[2][ctx2]);
 1440|   203k|                } else {
 1441|   203k|                    const int ctx2 = av1_get_fwd_ref_1_ctx(t->a, &t->l, by4, bx4,
 1442|   203k|                                                           have_top, have_left);
 1443|   203k|                    b->ref[0] = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   203k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1444|   203k|                                    ts->cdf.m.comp_fwd_ref[1][ctx2]);
 1445|   203k|                }
 1446|       |
 1447|       |                // second reference (bw)
 1448|   302k|                const int ctx3 = av1_get_bwd_ref_ctx(t->a, &t->l, by4, bx4,
 1449|   302k|                                                     have_top, have_left);
 1450|   302k|                if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   302k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1450:21): [True: 166k, False: 136k]
  ------------------
 1451|   302k|                        ts->cdf.m.comp_bwd_ref[0][ctx3]))
 1452|   166k|                {
 1453|   166k|                    b->ref[1] = 6;
 1454|   166k|                } else {
 1455|   136k|                    const int ctx4 = av1_get_bwd_ref_1_ctx(t->a, &t->l, by4, bx4,
 1456|   136k|                                                           have_top, have_left);
 1457|   136k|                    b->ref[1] = 4 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   136k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1458|   136k|                                        ts->cdf.m.comp_bwd_ref[1][ctx4]);
 1459|   136k|                }
 1460|   302k|            } else {
 1461|       |                // unidir
 1462|  69.7k|                const int uctx_p = av1_get_uni_p_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  280|  69.7k|#define av1_get_uni_p_ctx av1_get_ref_ctx
  ------------------
 1463|  69.7k|                                                     have_top, have_left);
 1464|  69.7k|                if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  69.7k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1464:21): [True: 17.5k, False: 52.2k]
  ------------------
 1465|  69.7k|                        ts->cdf.m.comp_uni_ref[0][uctx_p]))
 1466|  17.5k|                {
 1467|  17.5k|                    b->ref[0] = 4;
 1468|  17.5k|                    b->ref[1] = 6;
 1469|  52.2k|                } else {
 1470|  52.2k|                    const int uctx_p1 = av1_get_uni_p1_ctx(t->a, &t->l, by4, bx4,
 1471|  52.2k|                                                           have_top, have_left);
 1472|  52.2k|                    b->ref[0] = 0;
 1473|  52.2k|                    b->ref[1] = 1 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  52.2k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1474|  52.2k|                                        ts->cdf.m.comp_uni_ref[1][uctx_p1]);
 1475|  52.2k|                    if (b->ref[1] == 2) {
  ------------------
  |  Branch (1475:25): [True: 32.7k, False: 19.4k]
  ------------------
 1476|  32.7k|                        const int uctx_p2 = av1_get_uni_p2_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  281|  32.7k|#define av1_get_uni_p2_ctx av1_get_fwd_ref_2_ctx
  ------------------
 1477|  32.7k|                                                               have_top, have_left);
 1478|  32.7k|                        b->ref[1] += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  32.7k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1479|  32.7k|                                         ts->cdf.m.comp_uni_ref[2][uctx_p2]);
 1480|  32.7k|                    }
 1481|  52.2k|                }
 1482|  69.7k|            }
 1483|   372k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   372k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 372k]
  |  |  ------------------
  |  |   35|   372k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   372k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1484|      0|                printf("Post-refs[%d/%d]: r=%d\n",
 1485|      0|                       b->ref[0], b->ref[1], ts->msac.rng);
 1486|       |
 1487|   372k|            refmvs_candidate mvstack[8];
 1488|   372k|            int n_mvs, ctx;
 1489|   372k|            dav1d_refmvs_find(&t->rt, mvstack, &n_mvs, &ctx,
 1490|   372k|                              (union refmvs_refpair) { .ref = {
 1491|   372k|                                    b->ref[0] + 1, b->ref[1] + 1 }},
 1492|   372k|                              bs, intra_edge_flags, t->by, t->bx);
 1493|       |
 1494|   372k|            b->inter_mode = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   372k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 1495|   372k|                                ts->cdf.m.comp_inter_mode[ctx],
 1496|   372k|                                N_COMP_INTER_PRED_MODES - 1);
 1497|   372k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   372k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 372k]
  |  |  ------------------
  |  |   35|   372k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   372k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1498|      0|                printf("Post-compintermode[%d,ctx=%d,n_mvs=%d]: r=%d\n",
 1499|      0|                       b->inter_mode, ctx, n_mvs, ts->msac.rng);
 1500|       |
 1501|   372k|            const uint8_t *const im = dav1d_comp_inter_pred_modes[b->inter_mode];
 1502|   372k|            b->drl_idx = NEAREST_DRL;
 1503|   372k|            if (b->inter_mode == NEWMV_NEWMV) {
  ------------------
  |  Branch (1503:17): [True: 76.2k, False: 296k]
  ------------------
 1504|  76.2k|                if (n_mvs > 1) { // NEARER, NEAR or NEARISH
  ------------------
  |  Branch (1504:21): [True: 76.2k, False: 2]
  ------------------
 1505|  76.2k|                    const int drl_ctx_v1 = get_drl_context(mvstack, 0);
 1506|  76.2k|                    b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  76.2k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1507|  76.2k|                                      ts->cdf.m.drl_bit[drl_ctx_v1]);
 1508|  76.2k|                    if (b->drl_idx == NEARER_DRL && n_mvs > 2) {
  ------------------
  |  Branch (1508:25): [True: 47.9k, False: 28.2k]
  |  Branch (1508:53): [True: 9.04k, False: 38.9k]
  ------------------
 1509|  9.04k|                        const int drl_ctx_v2 = get_drl_context(mvstack, 1);
 1510|  9.04k|                        b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  9.04k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1511|  9.04k|                                          ts->cdf.m.drl_bit[drl_ctx_v2]);
 1512|  9.04k|                    }
 1513|  76.2k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  76.2k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 76.2k]
  |  |  ------------------
  |  |   35|  76.2k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  76.2k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1514|      0|                        printf("Post-drlidx[%d,n_mvs=%d]: r=%d\n",
 1515|      0|                               b->drl_idx, n_mvs, ts->msac.rng);
 1516|  76.2k|                }
 1517|   296k|            } else if (im[0] == NEARMV || im[1] == NEARMV) {
  ------------------
  |  Branch (1517:24): [True: 66.4k, False: 230k]
  |  Branch (1517:43): [True: 11.6k, False: 218k]
  ------------------
 1518|  78.3k|                b->drl_idx = NEARER_DRL;
 1519|  78.3k|                if (n_mvs > 2) { // NEAR or NEARISH
  ------------------
  |  Branch (1519:21): [True: 17.1k, False: 61.1k]
  ------------------
 1520|  17.1k|                    const int drl_ctx_v2 = get_drl_context(mvstack, 1);
 1521|  17.1k|                    b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  17.1k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1522|  17.1k|                                      ts->cdf.m.drl_bit[drl_ctx_v2]);
 1523|  17.1k|                    if (b->drl_idx == NEAR_DRL && n_mvs > 3) {
  ------------------
  |  Branch (1523:25): [True: 7.92k, False: 9.21k]
  |  Branch (1523:51): [True: 5.51k, False: 2.41k]
  ------------------
 1524|  5.51k|                        const int drl_ctx_v3 = get_drl_context(mvstack, 2);
 1525|  5.51k|                        b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  5.51k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1526|  5.51k|                                          ts->cdf.m.drl_bit[drl_ctx_v3]);
 1527|  5.51k|                    }
 1528|  17.1k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  17.1k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 17.1k]
  |  |  ------------------
  |  |   35|  17.1k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  17.1k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1529|      0|                        printf("Post-drlidx[%d,n_mvs=%d]: r=%d\n",
 1530|      0|                               b->drl_idx, n_mvs, ts->msac.rng);
 1531|  17.1k|                }
 1532|  78.3k|            }
 1533|   372k|            assert(b->drl_idx >= NEAREST_DRL && b->drl_idx <= NEARISH_DRL);
  ------------------
  |  Branch (1533:13): [True: 372k, False: 288]
  |  Branch (1533:13): [True: 372k, False: 18.4E]
  ------------------
 1534|       |
 1535|   372k|#define assign_comp_mv(idx) \
 1536|   372k|            switch (im[idx]) { \
 1537|   372k|            case NEARMV: \
 1538|   372k|            case NEARESTMV: \
 1539|   372k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
 1540|   372k|                fix_mv_precision(f->frame_hdr, &b->mv[idx]); \
 1541|   372k|                break; \
 1542|   372k|            case GLOBALMV: \
 1543|   372k|                has_subpel_filter |= \
 1544|   372k|                    f->frame_hdr->gmv[b->ref[idx]].type == DAV1D_WM_TYPE_TRANSLATION; \
 1545|   372k|                b->mv[idx] = get_gmv_2d(&f->frame_hdr->gmv[b->ref[idx]], \
 1546|   372k|                                        t->bx, t->by, bw4, bh4, f->frame_hdr); \
 1547|   372k|                break; \
 1548|   372k|            case NEWMV: \
 1549|   372k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
 1550|   372k|                const int mv_prec = f->frame_hdr->hp - f->frame_hdr->force_integer_mv; \
 1551|   372k|                read_mv_residual(ts, &b->mv[idx], mv_prec); \
 1552|   372k|                break; \
 1553|   372k|            }
 1554|   372k|            has_subpel_filter = imin(bw4, bh4) == 1 ||
  ------------------
  |  Branch (1554:33): [True: 110, False: 372k]
  ------------------
 1555|   372k|                                b->inter_mode != GLOBALMV_GLOBALMV;
  ------------------
  |  Branch (1555:33): [True: 332k, False: 40.2k]
  ------------------
 1556|   372k|            assign_comp_mv(0);
  ------------------
  |  | 1536|   372k|            switch (im[idx]) { \
  |  |  ------------------
  |  |  |  Branch (1536:21): [True: 372k, False: 18.4E]
  |  |  ------------------
  |  | 1537|  66.6k|            case NEARMV: \
  |  |  ------------------
  |  |  |  Branch (1537:13): [True: 66.6k, False: 305k]
  |  |  ------------------
  |  | 1538|   226k|            case NEARESTMV: \
  |  |  ------------------
  |  |  |  Branch (1538:13): [True: 159k, False: 212k]
  |  |  ------------------
  |  | 1539|   226k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
  |  | 1540|   226k|                fix_mv_precision(f->frame_hdr, &b->mv[idx]); \
  |  | 1541|   226k|                break; \
  |  | 1542|  66.6k|            case GLOBALMV: \
  |  |  ------------------
  |  |  |  Branch (1542:13): [True: 40.2k, False: 332k]
  |  |  ------------------
  |  | 1543|  40.2k|                has_subpel_filter |= \
  |  | 1544|  40.2k|                    f->frame_hdr->gmv[b->ref[idx]].type == DAV1D_WM_TYPE_TRANSLATION; \
  |  | 1545|  40.2k|                b->mv[idx] = get_gmv_2d(&f->frame_hdr->gmv[b->ref[idx]], \
  |  | 1546|  40.2k|                                        t->bx, t->by, bw4, bh4, f->frame_hdr); \
  |  | 1547|  40.2k|                break; \
  |  | 1548|   106k|            case NEWMV: \
  |  |  ------------------
  |  |  |  Branch (1548:13): [True: 106k, False: 265k]
  |  |  ------------------
  |  | 1549|   106k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
  |  | 1550|   106k|                const int mv_prec = f->frame_hdr->hp - f->frame_hdr->force_integer_mv; \
  |  | 1551|   106k|                read_mv_residual(ts, &b->mv[idx], mv_prec); \
  |  | 1552|   106k|                break; \
  |  | 1553|   372k|            }
  ------------------
 1557|   372k|            assign_comp_mv(1);
  ------------------
  |  | 1536|   372k|            switch (im[idx]) { \
  |  |  ------------------
  |  |  |  Branch (1536:21): [True: 373k, False: 18.4E]
  |  |  ------------------
  |  | 1537|  65.7k|            case NEARMV: \
  |  |  ------------------
  |  |  |  Branch (1537:13): [True: 65.7k, False: 306k]
  |  |  ------------------
  |  | 1538|   217k|            case NEARESTMV: \
  |  |  ------------------
  |  |  |  Branch (1538:13): [True: 151k, False: 221k]
  |  |  ------------------
  |  | 1539|   217k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
  |  | 1540|   217k|                fix_mv_precision(f->frame_hdr, &b->mv[idx]); \
  |  | 1541|   217k|                break; \
  |  | 1542|  65.7k|            case GLOBALMV: \
  |  |  ------------------
  |  |  |  Branch (1542:13): [True: 40.2k, False: 332k]
  |  |  ------------------
  |  | 1543|  40.2k|                has_subpel_filter |= \
  |  | 1544|  40.2k|                    f->frame_hdr->gmv[b->ref[idx]].type == DAV1D_WM_TYPE_TRANSLATION; \
  |  | 1545|  40.2k|                b->mv[idx] = get_gmv_2d(&f->frame_hdr->gmv[b->ref[idx]], \
  |  | 1546|  40.2k|                                        t->bx, t->by, bw4, bh4, f->frame_hdr); \
  |  | 1547|  40.2k|                break; \
  |  | 1548|   115k|            case NEWMV: \
  |  |  ------------------
  |  |  |  Branch (1548:13): [True: 115k, False: 256k]
  |  |  ------------------
  |  | 1549|   115k|                b->mv[idx] = mvstack[b->drl_idx].mv.mv[idx]; \
  |  | 1550|   115k|                const int mv_prec = f->frame_hdr->hp - f->frame_hdr->force_integer_mv; \
  |  | 1551|   115k|                read_mv_residual(ts, &b->mv[idx], mv_prec); \
  |  | 1552|   115k|                break; \
  |  | 1553|   372k|            }
  ------------------
 1558|   372k|#undef assign_comp_mv
 1559|   372k|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   372k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 372k]
  |  |  ------------------
  |  |   35|   372k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   372k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1560|      0|                printf("Post-residual_mv[1:y=%d,x=%d,2:y=%d,x=%d]: r=%d\n",
 1561|      0|                       b->mv[0].y, b->mv[0].x, b->mv[1].y, b->mv[1].x,
 1562|      0|                       ts->msac.rng);
 1563|       |
 1564|       |            // jnt_comp vs. seg vs. wedge
 1565|   372k|            int is_segwedge = 0;
 1566|   372k|            if (f->seq_hdr->masked_compound) {
  ------------------
  |  Branch (1566:17): [True: 295k, False: 77.3k]
  ------------------
 1567|   295k|                const int mask_ctx = get_mask_comp_ctx(t->a, &t->l, by4, bx4);
 1568|       |
 1569|   295k|                is_segwedge = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   295k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1570|   295k|                                  ts->cdf.m.mask_comp[mask_ctx]);
 1571|   295k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   295k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 295k]
  |  |  ------------------
  |  |   35|   295k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   295k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1572|      0|                    printf("Post-segwedge_vs_jntavg[%d,ctx=%d]: r=%d\n",
 1573|      0|                           is_segwedge, mask_ctx, ts->msac.rng);
 1574|   295k|            }
 1575|       |
 1576|   372k|            if (!is_segwedge) {
  ------------------
  |  Branch (1576:17): [True: 305k, False: 67.7k]
  ------------------
 1577|   305k|                if (f->seq_hdr->jnt_comp) {
  ------------------
  |  Branch (1577:21): [True: 228k, False: 76.4k]
  ------------------
 1578|   228k|                    const int jnt_ctx =
 1579|   228k|                        get_jnt_comp_ctx(f->seq_hdr->order_hint_n_bits,
 1580|   228k|                                         f->cur.frame_hdr->frame_offset,
 1581|   228k|                                         f->refp[b->ref[0]].p.frame_hdr->frame_offset,
 1582|   228k|                                         f->refp[b->ref[1]].p.frame_hdr->frame_offset,
 1583|   228k|                                         t->a, &t->l, by4, bx4);
 1584|   228k|                    b->comp_type = COMP_INTER_WEIGHTED_AVG +
 1585|   228k|                                   dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   228k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1586|   228k|                                       ts->cdf.m.jnt_comp[jnt_ctx]);
 1587|   228k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   228k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 228k]
  |  |  ------------------
  |  |   35|   228k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   228k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1588|      0|                        printf("Post-jnt_comp[%d,ctx=%d[ac:%d,ar:%d,lc:%d,lr:%d]]: r=%d\n",
 1589|      0|                               b->comp_type == COMP_INTER_AVG,
 1590|      0|                               jnt_ctx, t->a->comp_type[bx4], t->a->ref[0][bx4],
 1591|      0|                               t->l.comp_type[by4], t->l.ref[0][by4],
 1592|      0|                               ts->msac.rng);
 1593|   228k|                } else {
 1594|  76.4k|                    b->comp_type = COMP_INTER_AVG;
 1595|  76.4k|                }
 1596|   305k|            } else {
 1597|  67.7k|                if (wedge_allowed_mask & (1 << bs)) {
  ------------------
  |  Branch (1597:21): [True: 50.9k, False: 16.7k]
  ------------------
 1598|  50.9k|                    const int ctx = dav1d_wedge_ctx_lut[bs];
 1599|  50.9k|                    b->comp_type = COMP_INTER_WEDGE -
 1600|  50.9k|                                   dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  50.9k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1601|  50.9k|                                       ts->cdf.m.wedge_comp[ctx]);
 1602|  50.9k|                    if (b->comp_type == COMP_INTER_WEDGE)
  ------------------
  |  Branch (1602:25): [True: 20.2k, False: 30.6k]
  ------------------
 1603|  20.2k|                        b->wedge_idx = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|  20.2k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1604|  50.9k|                                           ts->cdf.m.wedge_idx[ctx], 15);
 1605|  50.9k|                } else {
 1606|  16.7k|                    b->comp_type = COMP_INTER_SEG;
 1607|  16.7k|                }
 1608|  67.7k|                b->mask_sign = dav1d_msac_decode_bool_equi(&ts->msac);
  ------------------
  |  |   53|  67.7k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
 1609|  67.7k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  67.7k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 67.7k]
  |  |  ------------------
  |  |   35|  67.7k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  67.7k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1610|      0|                    printf("Post-seg/wedge[%d,wedge_idx=%d,sign=%d]: r=%d\n",
 1611|      0|                           b->comp_type == COMP_INTER_WEDGE,
 1612|      0|                           b->wedge_idx, b->mask_sign, ts->msac.rng);
 1613|  67.7k|            }
 1614|  3.52M|        } else {
 1615|  3.52M|            b->comp_type = COMP_INTER_NONE;
 1616|       |
 1617|       |            // ref
 1618|  3.52M|            if (seg && seg->ref > 0) {
  ------------------
  |  Branch (1618:17): [True: 2.12M, False: 1.39M]
  |  Branch (1618:24): [True: 254k, False: 1.87M]
  ------------------
 1619|   254k|                b->ref[0] = seg->ref - 1;
 1620|  3.26M|            } else if (seg && (seg->globalmv || seg->skip)) {
  ------------------
  |  Branch (1620:24): [True: 1.87M, False: 1.39M]
  |  Branch (1620:32): [True: 1.71M, False: 163k]
  |  Branch (1620:49): [True: 23.9k, False: 139k]
  ------------------
 1621|  1.73M|                b->ref[0] = 0;
 1622|  1.73M|            } else {
 1623|  1.53M|                const int ctx1 = av1_get_ref_ctx(t->a, &t->l, by4, bx4,
 1624|  1.53M|                                                 have_top, have_left);
 1625|  1.53M|                if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  1.53M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1625:21): [True: 524k, False: 1.00M]
  ------------------
 1626|  1.53M|                                                 ts->cdf.m.ref[0][ctx1]))
 1627|   524k|                {
 1628|   524k|                    const int ctx2 = av1_get_ref_2_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  275|   524k|#define av1_get_ref_2_ctx av1_get_bwd_ref_ctx
  ------------------
 1629|   524k|                                                       have_top, have_left);
 1630|   524k|                    if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   524k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1630:25): [True: 375k, False: 149k]
  ------------------
 1631|   524k|                                                     ts->cdf.m.ref[1][ctx2]))
 1632|   375k|                    {
 1633|   375k|                        b->ref[0] = 6;
 1634|   375k|                    } else {
 1635|   149k|                        const int ctx3 = av1_get_ref_6_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  279|   149k|#define av1_get_ref_6_ctx av1_get_bwd_ref_1_ctx
  ------------------
 1636|   149k|                                                           have_top, have_left);
 1637|   149k|                        b->ref[0] = 4 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   149k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1638|   149k|                                            ts->cdf.m.ref[5][ctx3]);
 1639|   149k|                    }
 1640|  1.00M|                } else {
 1641|  1.00M|                    const int ctx2 = av1_get_ref_3_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  276|  1.00M|#define av1_get_ref_3_ctx av1_get_fwd_ref_ctx
  ------------------
 1642|  1.00M|                                                       have_top, have_left);
 1643|  1.00M|                    if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  1.00M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1643:25): [True: 146k, False: 861k]
  ------------------
 1644|  1.00M|                                                     ts->cdf.m.ref[2][ctx2]))
 1645|   146k|                    {
 1646|   146k|                        const int ctx3 = av1_get_ref_5_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  278|   146k|#define av1_get_ref_5_ctx av1_get_fwd_ref_2_ctx
  ------------------
 1647|   146k|                                                           have_top, have_left);
 1648|   146k|                        b->ref[0] = 2 + dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   146k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1649|   146k|                                            ts->cdf.m.ref[4][ctx3]);
 1650|   861k|                    } else {
 1651|   861k|                        const int ctx3 = av1_get_ref_4_ctx(t->a, &t->l, by4, bx4,
  ------------------
  |  |  277|   861k|#define av1_get_ref_4_ctx av1_get_fwd_ref_1_ctx
  ------------------
 1652|   861k|                                                           have_top, have_left);
 1653|   861k|                        b->ref[0] = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   861k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1654|   861k|                                        ts->cdf.m.ref[3][ctx3]);
 1655|   861k|                    }
 1656|  1.00M|                }
 1657|  1.53M|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.53M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.53M]
  |  |  ------------------
  |  |   35|  1.53M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.53M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1658|      0|                    printf("Post-ref[%d]: r=%d\n", b->ref[0], ts->msac.rng);
 1659|  1.53M|            }
 1660|  3.52M|            b->ref[1] = -1;
 1661|       |
 1662|  3.52M|            refmvs_candidate mvstack[8];
 1663|  3.52M|            int n_mvs, ctx;
 1664|  3.52M|            dav1d_refmvs_find(&t->rt, mvstack, &n_mvs, &ctx,
 1665|  3.52M|                              (union refmvs_refpair) { .ref = { b->ref[0] + 1, -1 }},
 1666|  3.52M|                              bs, intra_edge_flags, t->by, t->bx);
 1667|       |
 1668|       |            // mode parsing and mv derivation from ref_mvs
 1669|  3.52M|            if ((seg && (seg->skip || seg->globalmv)) ||
  ------------------
  |  Branch (1669:18): [True: 2.13M, False: 1.39M]
  |  Branch (1669:26): [True: 1.97M, False: 154k]
  |  Branch (1669:39): [True: 7.70k, False: 146k]
  ------------------
 1670|  1.55M|                dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  1.55M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1670:17): [True: 1.00M, False: 556k]
  ------------------
 1671|  1.55M|                                             ts->cdf.m.newmv_mode[ctx & 7]))
 1672|  2.98M|            {
 1673|  2.98M|                if ((seg && (seg->skip || seg->globalmv)) ||
  ------------------
  |  Branch (1673:22): [True: 2.08M, False: 907k]
  |  Branch (1673:30): [True: 1.97M, False: 103k]
  |  Branch (1673:43): [True: 7.71k, False: 95.3k]
  ------------------
 1674|  1.00M|                    !dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  1.00M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1674:21): [True: 77.2k, False: 925k]
  ------------------
 1675|  1.00M|                         ts->cdf.m.globalmv_mode[(ctx >> 3) & 1]))
 1676|  2.06M|                {
 1677|  2.06M|                    b->inter_mode = GLOBALMV;
 1678|  2.06M|                    b->mv[0] = get_gmv_2d(&f->frame_hdr->gmv[b->ref[0]],
 1679|  2.06M|                                          t->bx, t->by, bw4, bh4, f->frame_hdr);
 1680|  2.06M|                    has_subpel_filter = imin(bw4, bh4) == 1 ||
  ------------------
  |  Branch (1680:41): [True: 129k, False: 1.93M]
  ------------------
 1681|  1.93M|                        f->frame_hdr->gmv[b->ref[0]].type == DAV1D_WM_TYPE_TRANSLATION;
  ------------------
  |  Branch (1681:25): [True: 42.2k, False: 1.89M]
  ------------------
 1682|  2.06M|                } else {
 1683|   925k|                    has_subpel_filter = 1;
 1684|   925k|                    if (dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   925k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1684:25): [True: 402k, False: 523k]
  ------------------
 1685|   925k|                            ts->cdf.m.refmv_mode[(ctx >> 4) & 15]))
 1686|   402k|                    { // NEAREST, NEARER, NEAR or NEARISH
 1687|   402k|                        b->inter_mode = NEARMV;
 1688|   402k|                        b->drl_idx = NEARER_DRL;
 1689|   402k|                        if (n_mvs > 2) { // NEARER, NEAR or NEARISH
  ------------------
  |  Branch (1689:29): [True: 160k, False: 241k]
  ------------------
 1690|   160k|                            const int drl_ctx_v2 = get_drl_context(mvstack, 1);
 1691|   160k|                            b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   160k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1692|   160k|                                              ts->cdf.m.drl_bit[drl_ctx_v2]);
 1693|   160k|                            if (b->drl_idx == NEAR_DRL && n_mvs > 3) { // NEAR or NEARISH
  ------------------
  |  Branch (1693:33): [True: 84.9k, False: 75.9k]
  |  Branch (1693:59): [True: 49.3k, False: 35.5k]
  ------------------
 1694|  49.3k|                                const int drl_ctx_v3 =
 1695|  49.3k|                                    get_drl_context(mvstack, 2);
 1696|  49.3k|                                b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  49.3k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1697|  49.3k|                                                  ts->cdf.m.drl_bit[drl_ctx_v3]);
 1698|  49.3k|                            }
 1699|   160k|                        }
 1700|   523k|                    } else {
 1701|   523k|                        b->inter_mode = NEARESTMV;
 1702|   523k|                        b->drl_idx = NEAREST_DRL;
 1703|   523k|                    }
 1704|   925k|                    assert(b->drl_idx >= NEAREST_DRL && b->drl_idx <= NEARISH_DRL);
  ------------------
  |  Branch (1704:21): [True: 926k, False: 18.4E]
  |  Branch (1704:21): [True: 926k, False: 18.4E]
  ------------------
 1705|   926k|                    b->mv[0] = mvstack[b->drl_idx].mv.mv[0];
 1706|   926k|                    if (b->drl_idx < NEAR_DRL)
  ------------------
  |  Branch (1706:25): [True: 841k, False: 84.1k]
  ------------------
 1707|   841k|                        fix_mv_precision(f->frame_hdr, &b->mv[0]);
 1708|   926k|                }
 1709|       |
 1710|  2.98M|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  2.98M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 2.98M]
  |  |  ------------------
  |  |   35|  2.98M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  2.98M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1711|      0|                    printf("Post-intermode[%d,drl=%d,mv=y:%d,x:%d,n_mvs=%d]: r=%d\n",
 1712|      0|                           b->inter_mode, b->drl_idx, b->mv[0].y, b->mv[0].x, n_mvs,
 1713|      0|                           ts->msac.rng);
 1714|  2.98M|            } else {
 1715|   535k|                has_subpel_filter = 1;
 1716|   535k|                b->inter_mode = NEWMV;
 1717|   535k|                b->drl_idx = NEAREST_DRL;
 1718|   535k|                if (n_mvs > 1) { // NEARER, NEAR or NEARISH
  ------------------
  |  Branch (1718:21): [True: 440k, False: 94.4k]
  ------------------
 1719|   440k|                    const int drl_ctx_v1 = get_drl_context(mvstack, 0);
 1720|   440k|                    b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   440k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1721|   440k|                                      ts->cdf.m.drl_bit[drl_ctx_v1]);
 1722|   440k|                    if (b->drl_idx == NEARER_DRL && n_mvs > 2) { // NEAR or NEARISH
  ------------------
  |  Branch (1722:25): [True: 149k, False: 290k]
  |  Branch (1722:53): [True: 94.1k, False: 55.5k]
  ------------------
 1723|  94.1k|                        const int drl_ctx_v2 = get_drl_context(mvstack, 1);
 1724|  94.1k|                        b->drl_idx += dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  94.1k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1725|  94.1k|                                          ts->cdf.m.drl_bit[drl_ctx_v2]);
 1726|  94.1k|                    }
 1727|   440k|                }
 1728|   535k|                assert(b->drl_idx >= NEAREST_DRL && b->drl_idx <= NEARISH_DRL);
  ------------------
  |  Branch (1728:17): [True: 556k, False: 18.4E]
  |  Branch (1728:17): [True: 556k, False: 18.4E]
  ------------------
 1729|   556k|                if (n_mvs > 1) {
  ------------------
  |  Branch (1729:21): [True: 440k, False: 115k]
  ------------------
 1730|   440k|                    b->mv[0] = mvstack[b->drl_idx].mv.mv[0];
 1731|   440k|                } else {
 1732|   115k|                    assert(!b->drl_idx);
  ------------------
  |  Branch (1732:21): [True: 115k, False: 18.4E]
  ------------------
 1733|   115k|                    b->mv[0] = mvstack[0].mv.mv[0];
 1734|   115k|                    fix_mv_precision(f->frame_hdr, &b->mv[0]);
 1735|   115k|                }
 1736|   556k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   556k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 556k]
  |  |  ------------------
  |  |   35|   556k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   556k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1737|      0|                    printf("Post-intermode[%d,drl=%d]: r=%d\n",
 1738|      0|                           b->inter_mode, b->drl_idx, ts->msac.rng);
 1739|   556k|                const int mv_prec = f->frame_hdr->hp - f->frame_hdr->force_integer_mv;
 1740|   556k|                read_mv_residual(ts, &b->mv[0], mv_prec);
 1741|   556k|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   556k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 556k]
  |  |  ------------------
  |  |   35|   556k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   556k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1742|      0|                    printf("Post-residualmv[mv=y:%d,x:%d]: r=%d\n",
 1743|      0|                           b->mv[0].y, b->mv[0].x, ts->msac.rng);
 1744|   556k|            }
 1745|       |
 1746|       |            // interintra flags
 1747|  3.54M|            const int ii_sz_grp = dav1d_ymode_size_context[bs];
 1748|  3.54M|            if (f->seq_hdr->inter_intra &&
  ------------------
  |  Branch (1748:17): [True: 3.10M, False: 436k]
  ------------------
 1749|  3.10M|                interintra_allowed_mask & (1 << bs) &&
  ------------------
  |  Branch (1749:17): [True: 732k, False: 2.37M]
  ------------------
 1750|   732k|                dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   732k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  |  Branch (1750:17): [True: 113k, False: 618k]
  ------------------
 1751|   732k|                                             ts->cdf.m.interintra[ii_sz_grp]))
 1752|   113k|            {
 1753|   113k|                b->interintra_mode = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   113k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1754|   113k|                                         ts->cdf.m.interintra_mode[ii_sz_grp],
 1755|   113k|                                         N_INTER_INTRA_PRED_MODES - 1);
 1756|   113k|                const int wedge_ctx = dav1d_wedge_ctx_lut[bs];
 1757|   113k|                b->interintra_type = INTER_INTRA_BLEND +
 1758|   113k|                                     dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   113k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1759|   113k|                                         ts->cdf.m.interintra_wedge[wedge_ctx]);
 1760|   113k|                if (b->interintra_type == INTER_INTRA_WEDGE)
  ------------------
  |  Branch (1760:21): [True: 29.4k, False: 84.4k]
  ------------------
 1761|  29.4k|                    b->wedge_idx = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|  29.4k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
 1762|   113k|                                       ts->cdf.m.wedge_idx[wedge_ctx], 15);
 1763|  3.43M|            } else {
 1764|  3.43M|                b->interintra_type = INTER_INTRA_NONE;
 1765|  3.43M|            }
 1766|  3.54M|            if (DEBUG_BLOCK_INFO && f->seq_hdr->inter_intra &&
  ------------------
  |  |   34|  3.54M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 3.54M]
  |  |  ------------------
  |  |   35|  3.54M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  3.54M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (1766:37): [True: 0, False: 0]
  ------------------
 1767|      0|                interintra_allowed_mask & (1 << bs))
  ------------------
  |  Branch (1767:17): [True: 0, False: 0]
  ------------------
 1768|      0|            {
 1769|      0|                printf("Post-interintra[t=%d,m=%d,w=%d]: r=%d\n",
 1770|      0|                       b->interintra_type, b->interintra_mode,
 1771|      0|                       b->wedge_idx, ts->msac.rng);
 1772|      0|            }
 1773|       |
 1774|       |            // motion variation
 1775|  3.54M|            if (f->frame_hdr->switchable_motion_mode &&
  ------------------
  |  Branch (1775:17): [True: 3.32M, False: 217k]
  ------------------
 1776|  3.32M|                b->interintra_type == INTER_INTRA_NONE && imin(bw4, bh4) >= 2 &&
  ------------------
  |  Branch (1776:17): [True: 3.21M, False: 108k]
  |  Branch (1776:59): [True: 2.64M, False: 574k]
  ------------------
 1777|       |                // is not warped global motion
 1778|  2.64M|                !(!f->frame_hdr->force_integer_mv && b->inter_mode == GLOBALMV &&
  ------------------
  |  Branch (1778:19): [True: 2.59M, False: 46.1k]
  |  Branch (1778:54): [True: 1.77M, False: 822k]
  ------------------
 1779|  1.77M|                  f->frame_hdr->gmv[b->ref[0]].type > DAV1D_WM_TYPE_TRANSLATION) &&
  ------------------
  |  Branch (1779:19): [True: 798k, False: 977k]
  ------------------
 1780|       |                // has overlappable neighbours
 1781|  1.84M|                ((have_left && findoddzero(&t->l.intra[by4 + 1], h4 >> 1)) ||
  ------------------
  |  Branch (1781:19): [True: 1.10M, False: 745k]
  |  Branch (1781:32): [True: 1.03M, False: 66.5k]
  ------------------
 1782|   811k|                 (have_top && findoddzero(&t->a->intra[bx4 + 1], w4 >> 1))))
  ------------------
  |  Branch (1782:19): [True: 792k, False: 19.4k]
  |  Branch (1782:31): [True: 774k, False: 17.7k]
  ------------------
 1783|  1.80M|            {
 1784|       |                // reaching here means the block allows obmc - check warp by
 1785|       |                // finding matching-ref blocks in top/left edges
 1786|  1.80M|                uint64_t mask[2] = { 0, 0 };
 1787|  1.80M|                find_matching_ref(t, intra_edge_flags, bw4, bh4, w4, h4,
 1788|  1.80M|                                  have_left, have_top, b->ref[0], mask);
 1789|  1.80M|                const int allow_warp = !f->svc[b->ref[0]][0].scale &&
  ------------------
  |  Branch (1789:40): [True: 1.57M, False: 235k]
  ------------------
 1790|  1.57M|                    !f->frame_hdr->force_integer_mv &&
  ------------------
  |  Branch (1790:21): [True: 1.55M, False: 16.0k]
  ------------------
 1791|  1.55M|                    f->frame_hdr->warp_motion && (mask[0] | mask[1]);
  ------------------
  |  Branch (1791:21): [True: 600k, False: 956k]
  |  Branch (1791:50): [True: 541k, False: 59.2k]
  ------------------
 1792|       |
 1793|  1.80M|                b->motion_mode = allow_warp ?
  ------------------
  |  Branch (1793:34): [True: 541k, False: 1.26M]
  ------------------
 1794|   541k|                    dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   541k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1795|   541k|                        ts->cdf.m.motion_mode[bs], 2) :
 1796|  1.80M|                    dav1d_msac_decode_bool_adapt(&ts->msac, ts->cdf.m.obmc[bs]);
  ------------------
  |  |   52|  1.26M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 1797|  1.80M|                if (b->motion_mode == MM_WARP) {
  ------------------
  |  Branch (1797:21): [True: 169k, False: 1.63M]
  ------------------
 1798|   169k|                    has_subpel_filter = 0;
 1799|   169k|                    derive_warpmv(t, bw4, bh4, mask, b->mv[0], &t->warpmv);
 1800|   169k|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
 1801|   169k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   169k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 169k]
  |  |  ------------------
  |  |   35|   169k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   169k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1802|      0|                        printf("[ %c%x %c%x %c%x\n  %c%x %c%x %c%x ]\n"
 1803|      0|                               "alpha=%c%x, beta=%c%x, gamma=%c%x, delta=%c%x, "
 1804|      0|                               "mv=y:%d,x:%d\n",
 1805|      0|                               signabs(t->warpmv.matrix[0]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1806|      0|                               signabs(t->warpmv.matrix[1]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1807|      0|                               signabs(t->warpmv.matrix[2]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1808|      0|                               signabs(t->warpmv.matrix[3]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1809|      0|                               signabs(t->warpmv.matrix[4]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1810|      0|                               signabs(t->warpmv.matrix[5]),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1811|      0|                               signabs(t->warpmv.u.p.alpha),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1812|      0|                               signabs(t->warpmv.u.p.beta),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1813|      0|                               signabs(t->warpmv.u.p.gamma),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1814|      0|                               signabs(t->warpmv.u.p.delta),
  ------------------
  |  | 1800|      0|#define signabs(v) v < 0 ? '-' : ' ', abs(v)
  |  |  ------------------
  |  |  |  Branch (1800:20): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1815|      0|                               b->mv[0].y, b->mv[0].x);
 1816|   169k|#undef signabs
 1817|   169k|                    if (t->frame_thread.pass) {
  ------------------
  |  Branch (1817:25): [True: 169k, False: 455]
  ------------------
 1818|   169k|                        if (t->warpmv.type == DAV1D_WM_TYPE_AFFINE) {
  ------------------
  |  Branch (1818:29): [True: 153k, False: 15.6k]
  ------------------
 1819|   153k|                            b->matrix[0] = t->warpmv.matrix[2] - 0x10000;
 1820|   153k|                            b->matrix[1] = t->warpmv.matrix[3];
 1821|   153k|                            b->matrix[2] = t->warpmv.matrix[4];
 1822|   153k|                            b->matrix[3] = t->warpmv.matrix[5] - 0x10000;
 1823|   153k|                        } else {
 1824|  15.6k|                            b->matrix[0] = INT16_MIN;
 1825|  15.6k|                        }
 1826|   169k|                    }
 1827|   169k|                }
 1828|       |
 1829|  1.80M|                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  1.80M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.80M]
  |  |  ------------------
  |  |   35|  1.80M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.80M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1830|      0|                    printf("Post-motionmode[%d]: r=%d [mask: 0x%" PRIx64 "/0x%"
 1831|      0|                           PRIx64 "]\n", b->motion_mode, ts->msac.rng, mask[0],
 1832|      0|                            mask[1]);
 1833|  1.80M|            } else {
 1834|  1.73M|                b->motion_mode = MM_TRANSLATION;
 1835|  1.73M|            }
 1836|  3.54M|        }
 1837|       |
 1838|       |        // subpel filter
 1839|  3.94M|        enum Dav1dFilterMode filter[2];
 1840|  3.94M|        if (f->frame_hdr->subpel_filter_mode == DAV1D_FILTER_SWITCHABLE) {
  ------------------
  |  Branch (1840:13): [True: 2.18M, False: 1.76M]
  ------------------
 1841|  2.18M|            if (has_subpel_filter) {
  ------------------
  |  Branch (1841:17): [True: 910k, False: 1.27M]
  ------------------
 1842|   910k|                const int comp = b->comp_type != COMP_INTER_NONE;
 1843|   910k|                const int ctx1 = get_filter_ctx(t->a, &t->l, comp, 0, b->ref[0],
 1844|   910k|                                                by4, bx4);
 1845|   910k|                filter[0] = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   910k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1846|   910k|                               ts->cdf.m.filter[0][ctx1],
 1847|   910k|                               DAV1D_N_SWITCHABLE_FILTERS - 1);
 1848|   910k|                if (f->seq_hdr->dual_filter) {
  ------------------
  |  Branch (1848:21): [True: 452k, False: 458k]
  ------------------
 1849|   452k|                    const int ctx2 = get_filter_ctx(t->a, &t->l, comp, 1,
 1850|   452k|                                                    b->ref[0], by4, bx4);
 1851|   452k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   452k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 452k]
  |  |  ------------------
  |  |   35|   452k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   452k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1852|      0|                        printf("Post-subpel_filter1[%d,ctx=%d]: r=%d\n",
 1853|      0|                               filter[0], ctx1, ts->msac.rng);
 1854|   452k|                    filter[1] = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   452k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 1855|   452k|                                    ts->cdf.m.filter[1][ctx2],
 1856|   452k|                                    DAV1D_N_SWITCHABLE_FILTERS - 1);
 1857|   452k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   452k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 452k]
  |  |  ------------------
  |  |   35|   452k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   452k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1858|      0|                        printf("Post-subpel_filter2[%d,ctx=%d]: r=%d\n",
 1859|      0|                               filter[1], ctx2, ts->msac.rng);
 1860|   458k|                } else {
 1861|   458k|                    filter[1] = filter[0];
 1862|   458k|                    if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   458k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 458k]
  |  |  ------------------
  |  |   35|   458k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   458k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1863|      0|                        printf("Post-subpel_filter[%d,ctx=%d]: r=%d\n",
 1864|      0|                               filter[0], ctx1, ts->msac.rng);
 1865|   458k|                }
 1866|  1.27M|            } else {
 1867|  1.27M|                filter[0] = filter[1] = DAV1D_FILTER_8TAP_REGULAR;
 1868|  1.27M|            }
 1869|  2.18M|        } else {
 1870|  1.76M|            filter[0] = filter[1] = f->frame_hdr->subpel_filter_mode;
 1871|  1.76M|        }
 1872|  3.94M|        b->filter2d = dav1d_filter_2d[filter[1]][filter[0]];
 1873|       |
 1874|  3.94M|        read_vartx_tree(t, b, bs, bx4, by4);
 1875|       |
 1876|       |        // reconstruction
 1877|  3.94M|        if (t->frame_thread.pass == 1) {
  ------------------
  |  Branch (1877:13): [True: 3.94M, False: 1.83k]
  ------------------
 1878|  3.94M|            f->bd_fn.read_coef_blocks(t, bs, b);
 1879|  3.94M|        } else {
 1880|  1.83k|            if (f->bd_fn.recon_b_inter(t, bs, b)) return -1;
  ------------------
  |  Branch (1880:17): [True: 0, False: 1.83k]
  ------------------
 1881|  1.83k|        }
 1882|       |
 1883|  3.94M|        if (f->frame_hdr->loopfilter.level_y[0] ||
  ------------------
  |  Branch (1883:13): [True: 3.41M, False: 525k]
  ------------------
 1884|   525k|            f->frame_hdr->loopfilter.level_y[1])
  ------------------
  |  Branch (1884:13): [True: 222k, False: 302k]
  ------------------
 1885|  3.63M|        {
 1886|  3.63M|            const int is_globalmv =
 1887|  3.63M|                b->inter_mode == (is_comp ? GLOBALMV_GLOBALMV : GLOBALMV);
  ------------------
  |  Branch (1887:35): [True: 330k, False: 3.30M]
  ------------------
 1888|  3.63M|            const uint8_t (*const lf_lvls)[8][2] = (const uint8_t (*)[8][2])
 1889|  3.63M|                &ts->lflvl[b->seg_id][0][b->ref[0] + 1][!is_globalmv];
 1890|  3.63M|            const uint16_t tx_split[2] = { b->tx_split0, b->tx_split1 };
 1891|  3.63M|            enum RectTxfmSize ytx = b->max_ytx, uvtx = b->uvtx;
 1892|  3.63M|            if (f->frame_hdr->segmentation.lossless[b->seg_id]) {
  ------------------
  |  Branch (1892:17): [True: 29.6k, False: 3.60M]
  ------------------
 1893|  29.6k|                ytx  = (enum RectTxfmSize) TX_4X4;
 1894|  29.6k|                uvtx = (enum RectTxfmSize) TX_4X4;
 1895|  29.6k|            }
 1896|  3.63M|            dav1d_create_lf_mask_inter(t->lf_mask, f->lf.level, f->b4_stride, lf_lvls,
 1897|  3.63M|                                       t->bx, t->by, f->w4, f->h4, b->skip, bs,
 1898|  3.63M|                                       ytx, tx_split, uvtx, f->cur.p.layout,
 1899|  3.63M|                                       &t->a->tx_lpf_y[bx4], &t->l.tx_lpf_y[by4],
 1900|  3.63M|                                       has_chroma ? &t->a->tx_lpf_uv[cbx4] : NULL,
  ------------------
  |  Branch (1900:40): [True: 1.11M, False: 2.51M]
  ------------------
 1901|  3.63M|                                       has_chroma ? &t->l.tx_lpf_uv[cby4] : NULL);
  ------------------
  |  Branch (1901:40): [True: 1.11M, False: 2.51M]
  ------------------
 1902|  3.63M|        }
 1903|       |
 1904|       |        // context updates
 1905|  3.94M|        if (is_comp)
  ------------------
  |  Branch (1905:13): [True: 398k, False: 3.54M]
  ------------------
 1906|   398k|            splat_tworef_mv(f->c, t, bs, b, bw4, bh4);
 1907|  3.54M|        else
 1908|  3.54M|            splat_oneref_mv(f->c, t, bs, b, bw4, bh4);
 1909|  3.94M|        BlockContext *edge = t->a;
 1910|  11.7M|        for (int i = 0, off = bx4; i < 2; i++, off = by4, edge = &t->l) {
  ------------------
  |  Branch (1910:36): [True: 7.84M, False: 3.94M]
  ------------------
 1911|  7.84M|#define set_ctx(rep_macro) \
 1912|  7.84M|            rep_macro(edge->seg_pred, off, seg_pred); \
 1913|  7.84M|            rep_macro(edge->skip_mode, off, b->skip_mode); \
 1914|  7.84M|            rep_macro(edge->intra, off, 0); \
 1915|  7.84M|            rep_macro(edge->skip, off, b->skip); \
 1916|  7.84M|            rep_macro(edge->pal_sz, off, 0); \
 1917|       |            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
 1918|  7.84M|            rep_macro(t->pal_sz_uv[i], off, 0); \
 1919|  7.84M|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
 1920|  7.84M|            rep_macro(edge->comp_type, off, b->comp_type); \
 1921|  7.84M|            rep_macro(edge->filter[0], off, filter[0]); \
 1922|  7.84M|            rep_macro(edge->filter[1], off, filter[1]); \
 1923|  7.84M|            rep_macro(edge->mode, off, b->inter_mode); \
 1924|  7.84M|            rep_macro(edge->ref[0], off, b->ref[0]); \
 1925|  7.84M|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
 1926|  7.84M|            case_set(b_dim[2 + i]);
  ------------------
  |  |   70|  7.84M|    switch (var) { \
  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 1912|   780k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|   780k|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|   780k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|   780k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|   780k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|   780k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|   780k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|   780k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|   780k|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|   780k|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|   780k|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|   780k|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|   780k|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|   780k|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   780k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   780k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 780k, False: 7.06M]
  |  |  ------------------
  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 1912|  1.46M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|  1.46M|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|  1.46M|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|  1.46M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|  1.46M|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|  1.46M|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|  1.46M|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|  1.46M|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|  1.46M|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|  1.46M|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|  1.46M|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|  1.46M|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|  1.46M|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|  1.46M|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|  1.46M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.46M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 1.46M, False: 6.37M]
  |  |  ------------------
  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 1912|  1.26M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|  1.26M|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|  1.26M|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|  1.26M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|  1.26M|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|  1.26M|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|  1.26M|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|  1.26M|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|  1.26M|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|  1.26M|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|  1.26M|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|  1.26M|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|  1.26M|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|  1.26M|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.26M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.26M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 1.26M, False: 6.58M]
  |  |  ------------------
  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 1912|   512k|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|   512k|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|   512k|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|   512k|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|   512k|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|   512k|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|   512k|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|   512k|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|   512k|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|   512k|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|   512k|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|   512k|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|   512k|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|   512k|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   512k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   512k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 512k, False: 7.33M]
  |  |  ------------------
  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 1912|  2.38M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|  2.38M|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|  2.38M|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|  2.38M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|  2.38M|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|  2.38M|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|  2.38M|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|  2.38M|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|  2.38M|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|  2.38M|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|  2.38M|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|  2.38M|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|  2.38M|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|  2.38M|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  2.38M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.38M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.38M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.38M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.38M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 2.38M, False: 5.45M]
  |  |  ------------------
  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  | 1912|  1.44M|            rep_macro(edge->seg_pred, off, seg_pred); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1913|  1.44M|            rep_macro(edge->skip_mode, off, b->skip_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1914|  1.44M|            rep_macro(edge->intra, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1915|  1.44M|            rep_macro(edge->skip, off, b->skip); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1916|  1.44M|            rep_macro(edge->pal_sz, off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1917|  1.44M|            /* see aomedia bug 2183 for why this is outside if (has_chroma) */ \
  |  |  |  | 1918|  1.44M|            rep_macro(t->pal_sz_uv[i], off, 0); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1919|  1.44M|            rep_macro(edge->tx_intra, off, b_dim[2 + i]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1920|  1.44M|            rep_macro(edge->comp_type, off, b->comp_type); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1921|  1.44M|            rep_macro(edge->filter[0], off, filter[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1922|  1.44M|            rep_macro(edge->filter[1], off, filter[1]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1923|  1.44M|            rep_macro(edge->mode, off, b->inter_mode); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1924|  1.44M|            rep_macro(edge->ref[0], off, b->ref[0]); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1925|  1.44M|            rep_macro(edge->ref[1], off, ((uint8_t) b->ref[1]))
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  1.44M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  1.44M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  1.44M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  1.44M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 1.44M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 1.44M, False: 6.39M]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 7.84M]
  |  |  ------------------
  |  |   78|  7.84M|    }
  ------------------
  |  Branch (1926:13): [Folded, False: 0]
  ------------------
 1927|  7.84M|#undef set_ctx
 1928|  7.84M|        }
 1929|  3.94M|        if (has_chroma) {
  ------------------
  |  Branch (1929:13): [True: 1.35M, False: 2.58M]
  ------------------
 1930|  1.35M|            dav1d_memset_pow2[ulog2(cbw4)](&t->a->uvmode[cbx4], DC_PRED);
 1931|  1.35M|            dav1d_memset_pow2[ulog2(cbh4)](&t->l.uvmode[cby4], DC_PRED);
 1932|  1.35M|        }
 1933|  3.94M|    }
 1934|       |
 1935|       |    // update contexts
 1936|  11.8M|    if (f->frame_hdr->segmentation.enabled &&
  ------------------
  |  Branch (1936:9): [True: 3.46M, False: 8.34M]
  ------------------
 1937|  3.46M|        f->frame_hdr->segmentation.update_map)
  ------------------
  |  Branch (1937:9): [True: 2.56M, False: 905k]
  ------------------
 1938|  2.56M|    {
 1939|  2.56M|        uint8_t *seg_ptr = &f->cur_segmap[t->by * f->b4_stride + t->bx];
 1940|  2.56M|#define set_ctx(rep_macro) \
 1941|  2.56M|        for (int y = 0; y < bh4; y++) { \
 1942|  2.56M|            rep_macro(seg_ptr, 0, b->seg_id); \
 1943|  2.56M|            seg_ptr += f->b4_stride; \
 1944|  2.56M|        }
 1945|  2.56M|        case_set(b_dim[2]);
  ------------------
  |  |   70|  2.56M|    switch (var) { \
  |  |   71|   163k|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  | 1941|   488k|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 324k, False: 163k]
  |  |  |  |  ------------------
  |  |  |  | 1942|   324k|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   71|   324k|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|   324k|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|   324k|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|   324k|        }
  |  |  ------------------
  |  |  |  Branch (71:5): [True: 163k, False: 2.40M]
  |  |  ------------------
  |  |   72|   297k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  | 1941|  1.16M|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 864k, False: 297k]
  |  |  |  |  ------------------
  |  |  |  | 1942|   864k|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   72|   864k|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|   864k|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|   864k|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|   864k|        }
  |  |  ------------------
  |  |  |  Branch (72:5): [True: 297k, False: 2.26M]
  |  |  ------------------
  |  |   73|   382k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  | 1941|  2.13M|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 1.75M, False: 382k]
  |  |  |  |  ------------------
  |  |  |  | 1942|  1.75M|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   73|  1.75M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  1.75M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|  1.75M|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|  1.75M|        }
  |  |  ------------------
  |  |  |  Branch (73:5): [True: 382k, False: 2.18M]
  |  |  ------------------
  |  |   74|   145k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  | 1941|   992k|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 846k, False: 145k]
  |  |  |  |  ------------------
  |  |  |  | 1942|   846k|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   74|   846k|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|   846k|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|   846k|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|   846k|        }
  |  |  ------------------
  |  |  |  Branch (74:5): [True: 145k, False: 2.41M]
  |  |  ------------------
  |  |   75|   722k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  | 1941|  12.2M|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 11.4M, False: 722k]
  |  |  |  |  ------------------
  |  |  |  | 1942|  11.4M|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   75|  11.4M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  11.4M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  11.4M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  11.4M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 11.4M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|  11.4M|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|  11.4M|        }
  |  |  ------------------
  |  |  |  Branch (75:5): [True: 722k, False: 1.84M]
  |  |  ------------------
  |  |   76|   854k|    case 5: set_ctx(set_ctx32); break; \
  |  |  ------------------
  |  |  |  | 1941|  28.0M|        for (int y = 0; y < bh4; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (1941:25): [True: 27.2M, False: 854k]
  |  |  |  |  ------------------
  |  |  |  | 1942|  27.2M|            rep_macro(seg_ptr, 0, b->seg_id); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   76|  27.2M|    case 5: set_ctx(set_ctx32); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   66|  27.2M|#define set_ctx32(var, off, val) do { \
  |  |  |  |  |  |  |  |   67|  27.2M|        memset(&(var)[off], val, 32); \
  |  |  |  |  |  |  |  |   68|  27.2M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (68:14): [Folded, False: 27.2M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  | 1943|  27.2M|            seg_ptr += f->b4_stride; \
  |  |  |  | 1944|  27.2M|        }
  |  |  ------------------
  |  |  |  Branch (76:5): [True: 854k, False: 1.70M]
  |  |  ------------------
  |  |   77|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (77:5): [True: 0, False: 2.56M]
  |  |  ------------------
  |  |   78|  2.56M|    }
  ------------------
  |  Branch (1945:9): [Folded, False: 0]
  ------------------
 1946|  2.56M|#undef set_ctx
 1947|  2.56M|    }
 1948|  11.8M|    if (!b->skip) {
  ------------------
  |  Branch (1948:9): [True: 6.60M, False: 5.21M]
  ------------------
 1949|  6.60M|        uint16_t (*noskip_mask)[2] = &t->lf_mask->noskip_mask[by4 >> 1];
 1950|  6.60M|        const unsigned mask = (~0U >> (32 - bw4)) << (bx4 & 15);
 1951|  6.60M|        const int bx_idx = (bx4 & 16) >> 4;
 1952|  22.9M|        for (int y = 0; y < bh4; y += 2, noskip_mask++) {
  ------------------
  |  Branch (1952:25): [True: 16.3M, False: 6.60M]
  ------------------
 1953|  16.3M|            (*noskip_mask)[bx_idx] |= mask;
 1954|  16.3M|            if (bw4 == 32) // this should be mask >> 16, but it's 0xffffffff anyway
  ------------------
  |  Branch (1954:17): [True: 2.96M, False: 13.3M]
  ------------------
 1955|  2.96M|                (*noskip_mask)[1] |= mask;
 1956|  16.3M|        }
 1957|  6.60M|    }
 1958|       |
 1959|  11.8M|    if (t->frame_thread.pass == 1 && !b->intra && IS_INTER_OR_SWITCH(f->frame_hdr)) {
  ------------------
  |  |   36|  5.26M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 3.93M, False: 1.32M]
  |  |  ------------------
  ------------------
  |  Branch (1959:9): [True: 11.8M, False: 18.4E]
  |  Branch (1959:38): [True: 5.26M, False: 6.57M]
  ------------------
 1960|  3.93M|        const int sby = (t->by - ts->tiling.row_start) >> f->sb_shift;
 1961|  3.93M|        int (*const lowest_px)[2] = ts->lowest_pixel[sby];
 1962|       |
 1963|       |        // keep track of motion vectors for each reference
 1964|  3.93M|        if (b->comp_type == COMP_INTER_NONE) {
  ------------------
  |  Branch (1964:13): [True: 3.54M, False: 396k]
  ------------------
 1965|       |            // y
 1966|  3.54M|            if (imin(bw4, bh4) > 1 &&
  ------------------
  |  Branch (1966:17): [True: 2.90M, False: 638k]
  ------------------
 1967|  2.90M|                ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1967:19): [True: 1.92M, False: 974k]
  |  Branch (1967:48): [True: 753k, False: 1.17M]
  ------------------
 1968|  2.15M|                 (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1968:19): [True: 169k, False: 1.98M]
  |  Branch (1968:48): [True: 154k, False: 15.6k]
  ------------------
 1969|   907k|            {
 1970|   907k|                affine_lowest_px_luma(t, &lowest_px[b->ref[0]][0], b_dim,
 1971|   907k|                                      b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1971:39): [True: 154k, False: 753k]
  ------------------
 1972|   907k|                                      &f->frame_hdr->gmv[b->ref[0]]);
 1973|  2.63M|            } else {
 1974|  2.63M|                mc_lowest_px(&lowest_px[b->ref[0]][0], t->by, bh4, b->mv[0].y,
 1975|  2.63M|                             0, &f->svc[b->ref[0]][1]);
 1976|  2.63M|                if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1976:21): [True: 421k, False: 2.21M]
  ------------------
 1977|   421k|                    obmc_lowest_px(t, lowest_px, 0, b_dim, bx4, by4, w4, h4);
 1978|   421k|                }
 1979|  2.63M|            }
 1980|       |
 1981|       |            // uv
 1982|  3.54M|            if (has_chroma) {
  ------------------
  |  Branch (1982:17): [True: 1.19M, False: 2.34M]
  ------------------
 1983|       |                // sub8x8 derivation
 1984|  1.19M|                int is_sub8x8 = bw4 == ss_hor || bh4 == ss_ver;
  ------------------
  |  Branch (1984:33): [True: 61.2k, False: 1.13M]
  |  Branch (1984:50): [True: 48.2k, False: 1.08M]
  ------------------
 1985|  1.19M|                refmvs_block *const *r;
 1986|  1.19M|                if (is_sub8x8) {
  ------------------
  |  Branch (1986:21): [True: 109k, False: 1.08M]
  ------------------
 1987|   109k|                    assert(ss_hor == 1);
  ------------------
  |  Branch (1987:21): [True: 109k, False: 18.4E]
  ------------------
 1988|   109k|                    r = &t->rt.r[(t->by & 31) + 5];
 1989|   109k|                    if (bw4 == 1) is_sub8x8 &= r[0][t->bx - 1].ref.ref[0] > 0;
  ------------------
  |  Branch (1989:25): [True: 61.4k, False: 48.4k]
  ------------------
 1990|   109k|                    if (bh4 == ss_ver) is_sub8x8 &= r[-1][t->bx].ref.ref[0] > 0;
  ------------------
  |  Branch (1990:25): [True: 62.0k, False: 47.8k]
  ------------------
 1991|   109k|                    if (bw4 == 1 && bh4 == ss_ver)
  ------------------
  |  Branch (1991:25): [True: 61.4k, False: 48.4k]
  |  Branch (1991:37): [True: 13.6k, False: 47.8k]
  ------------------
 1992|  13.6k|                        is_sub8x8 &= r[-1][t->bx - 1].ref.ref[0] > 0;
 1993|   109k|                }
 1994|       |
 1995|       |                // chroma prediction
 1996|  1.19M|                if (is_sub8x8) {
  ------------------
  |  Branch (1996:21): [True: 96.5k, False: 1.10M]
  ------------------
 1997|  96.5k|                    assert(ss_hor == 1);
  ------------------
  |  Branch (1997:21): [True: 96.5k, False: 0]
  ------------------
 1998|  96.5k|                    if (bw4 == 1 && bh4 == ss_ver) {
  ------------------
  |  Branch (1998:25): [True: 54.1k, False: 42.3k]
  |  Branch (1998:37): [True: 11.2k, False: 42.9k]
  ------------------
 1999|  11.2k|                        const refmvs_block *const rr = &r[-1][t->bx - 1];
 2000|  11.2k|                        mc_lowest_px(&lowest_px[rr->ref.ref[0] - 1][1],
 2001|  11.2k|                                     t->by - 1, bh4, rr->mv.mv[0].y, ss_ver,
 2002|  11.2k|                                     &f->svc[rr->ref.ref[0] - 1][1]);
 2003|  11.2k|                    }
 2004|  96.5k|                    if (bw4 == 1) {
  ------------------
  |  Branch (2004:25): [True: 54.1k, False: 42.3k]
  ------------------
 2005|  54.1k|                        const refmvs_block *const rr = &r[0][t->bx - 1];
 2006|  54.1k|                        mc_lowest_px(&lowest_px[rr->ref.ref[0] - 1][1],
 2007|  54.1k|                                     t->by, bh4, rr->mv.mv[0].y, ss_ver,
 2008|  54.1k|                                     &f->svc[rr->ref.ref[0] - 1][1]);
 2009|  54.1k|                    }
 2010|  96.5k|                    if (bh4 == ss_ver) {
  ------------------
  |  Branch (2010:25): [True: 53.5k, False: 42.9k]
  ------------------
 2011|  53.5k|                        const refmvs_block *const rr = &r[-1][t->bx];
 2012|  53.5k|                        mc_lowest_px(&lowest_px[rr->ref.ref[0] - 1][1],
 2013|  53.5k|                                     t->by - 1, bh4, rr->mv.mv[0].y, ss_ver,
 2014|  53.5k|                                     &f->svc[rr->ref.ref[0] - 1][1]);
 2015|  53.5k|                    }
 2016|  96.5k|                    mc_lowest_px(&lowest_px[b->ref[0]][1], t->by, bh4,
 2017|  96.5k|                                 b->mv[0].y, ss_ver, &f->svc[b->ref[0]][1]);
 2018|  1.10M|                } else {
 2019|  1.10M|                    if (imin(cbw4, cbh4) > 1 &&
  ------------------
  |  Branch (2019:25): [True: 636k, False: 465k]
  ------------------
 2020|   636k|                        ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (2020:27): [True: 149k, False: 487k]
  |  Branch (2020:56): [True: 24.5k, False: 125k]
  ------------------
 2021|   612k|                         (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (2021:27): [True: 64.2k, False: 548k]
  |  Branch (2021:56): [True: 61.9k, False: 2.22k]
  ------------------
 2022|  86.5k|                    {
 2023|  86.5k|                        affine_lowest_px_chroma(t, &lowest_px[b->ref[0]][1], b_dim,
 2024|  86.5k|                                                b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (2024:49): [True: 61.9k, False: 24.5k]
  ------------------
 2025|  86.5k|                                                &f->frame_hdr->gmv[b->ref[0]]);
 2026|  1.01M|                    } else {
 2027|  1.01M|                        mc_lowest_px(&lowest_px[b->ref[0]][1],
 2028|  1.01M|                                     t->by & ~ss_ver, bh4 << (bh4 == ss_ver),
 2029|  1.01M|                                     b->mv[0].y, ss_ver, &f->svc[b->ref[0]][1]);
 2030|  1.01M|                        if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (2030:29): [True: 310k, False: 705k]
  ------------------
 2031|   310k|                            obmc_lowest_px(t, lowest_px, 1, b_dim, bx4, by4, w4, h4);
 2032|   310k|                        }
 2033|  1.01M|                    }
 2034|  1.10M|                }
 2035|  1.19M|            }
 2036|  3.54M|        } else {
 2037|       |            // y
 2038|  1.19M|            for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (2038:29): [True: 794k, False: 396k]
  ------------------
 2039|   794k|                if (b->inter_mode == GLOBALMV_GLOBALMV && f->gmv_warp_allowed[b->ref[i]]) {
  ------------------
  |  Branch (2039:21): [True: 80.5k, False: 714k]
  |  Branch (2039:59): [True: 10.9k, False: 69.5k]
  ------------------
 2040|  10.9k|                    affine_lowest_px_luma(t, &lowest_px[b->ref[i]][0], b_dim,
 2041|  10.9k|                                          &f->frame_hdr->gmv[b->ref[i]]);
 2042|   783k|                } else {
 2043|   783k|                    mc_lowest_px(&lowest_px[b->ref[i]][0], t->by, bh4,
 2044|   783k|                                 b->mv[i].y, 0, &f->svc[b->ref[i]][1]);
 2045|   783k|                }
 2046|   794k|            }
 2047|       |
 2048|       |            // uv
 2049|   476k|            if (has_chroma) for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (2049:17): [True: 158k, False: 237k]
  |  Branch (2049:45): [True: 317k, False: 158k]
  ------------------
 2050|   317k|                if (b->inter_mode == GLOBALMV_GLOBALMV &&
  ------------------
  |  Branch (2050:21): [True: 21.1k, False: 296k]
  ------------------
 2051|  21.1k|                    imin(cbw4, cbh4) > 1 && f->gmv_warp_allowed[b->ref[i]])
  ------------------
  |  Branch (2051:21): [True: 14.7k, False: 6.36k]
  |  Branch (2051:45): [True: 2.95k, False: 11.8k]
  ------------------
 2052|  2.95k|                {
 2053|  2.95k|                    affine_lowest_px_chroma(t, &lowest_px[b->ref[i]][1], b_dim,
 2054|  2.95k|                                            &f->frame_hdr->gmv[b->ref[i]]);
 2055|   314k|                } else {
 2056|   314k|                    mc_lowest_px(&lowest_px[b->ref[i]][1], t->by, bh4,
 2057|   314k|                                 b->mv[i].y, ss_ver, &f->svc[b->ref[i]][1]);
 2058|   314k|                }
 2059|   317k|            }
 2060|   396k|        }
 2061|  3.93M|    }
 2062|       |
 2063|  11.8M|    return 0;
 2064|  11.8M|}
decode.c:get_prev_frame_segid:
  499|   763k|{
  500|   763k|    assert(f->frame_hdr->primary_ref_frame != DAV1D_PRIMARY_REF_NONE);
  ------------------
  |  Branch (500:5): [True: 763k, False: 18.4E]
  ------------------
  501|       |
  502|   763k|    unsigned seg_id = 8;
  503|   763k|    ref_seg_map += by * stride + bx;
  504|   829k|    do {
  505|  11.5M|        for (int x = 0; x < w4; x++)
  ------------------
  |  Branch (505:25): [True: 10.7M, False: 829k]
  ------------------
  506|  10.7M|            seg_id = imin(seg_id, ref_seg_map[x]);
  507|   829k|        ref_seg_map += stride;
  508|   829k|    } while (--h4 > 0 && seg_id);
  ------------------
  |  Branch (508:14): [True: 799k, False: 29.6k]
  |  Branch (508:26): [True: 66.0k, False: 733k]
  ------------------
  509|   763k|    assert(seg_id < 8);
  ------------------
  |  Branch (509:5): [True: 760k, False: 2.87k]
  ------------------
  510|       |
  511|   760k|    return seg_id;
  512|   763k|}
decode.c:neg_deinterleave:
  169|  2.31M|static int neg_deinterleave(int diff, int ref, int max) {
  170|  2.31M|    if (!ref) return diff;
  ------------------
  |  Branch (170:9): [True: 1.96M, False: 354k]
  ------------------
  171|   354k|    if (ref >= (max - 1)) return max - diff - 1;
  ------------------
  |  Branch (171:9): [True: 83.7k, False: 270k]
  ------------------
  172|   270k|    if (2 * ref < max) {
  ------------------
  |  Branch (172:9): [True: 183k, False: 86.8k]
  ------------------
  173|   183k|        if (diff <= 2 * ref) {
  ------------------
  |  Branch (173:13): [True: 144k, False: 39.2k]
  ------------------
  174|   144k|            if (diff & 1)
  ------------------
  |  Branch (174:17): [True: 18.4k, False: 125k]
  ------------------
  175|  18.4k|                return ref + ((diff + 1) >> 1);
  176|   125k|            else
  177|   125k|                return ref - (diff >> 1);
  178|   144k|        }
  179|  39.2k|        return diff;
  180|   183k|    } else {
  181|  86.8k|        if (diff <= 2 * (max - ref - 1)) {
  ------------------
  |  Branch (181:13): [True: 72.8k, False: 13.9k]
  ------------------
  182|  72.8k|            if (diff & 1)
  ------------------
  |  Branch (182:17): [True: 13.8k, False: 59.0k]
  ------------------
  183|  13.8k|                return ref + ((diff + 1) >> 1);
  184|  59.0k|            else
  185|  59.0k|                return ref - (diff >> 1);
  186|  72.8k|        }
  187|  13.9k|        return max - (diff + 1);
  188|  86.8k|    }
  189|   270k|}
decode.c:read_pal_indices:
  419|   199k|{
  420|   199k|    Dav1dTileState *const ts = t->ts;
  421|   199k|    const ptrdiff_t stride = bw4 * 4;
  422|   199k|    assert(pal_idx);
  ------------------
  |  Branch (422:5): [True: 199k, False: 3]
  ------------------
  423|   199k|    uint8_t *const pal_tmp = t->scratch.pal_idx_uv;
  424|   199k|    pal_tmp[0] = dav1d_msac_decode_uniform(&ts->msac, pal_sz);
  425|   199k|    uint16_t (*const color_map_cdf)[8] =
  426|   199k|        ts->cdf.m.color_map[pl][pal_sz - 2];
  427|   199k|    uint8_t (*const order)[8] = t->scratch.pal_order;
  428|   199k|    uint8_t *const ctx = t->scratch.pal_ctx;
  429|  5.68M|    for (int i = 1; i < 4 * (w4 + h4) - 1; i++) {
  ------------------
  |  Branch (429:21): [True: 5.48M, False: 199k]
  ------------------
  430|       |        // top/left-to-bottom/right diagonals ("wave-front")
  431|  5.48M|        const int first = imin(i, w4 * 4 - 1);
  432|  5.48M|        const int last = imax(0, i - h4 * 4 + 1);
  433|  5.48M|        order_palette(pal_tmp, stride, i, first, last, order, ctx);
  434|  60.8M|        for (int j = first, m = 0; j >= last; j--, m++) {
  ------------------
  |  Branch (434:36): [True: 55.3M, False: 5.48M]
  ------------------
  435|  55.3M|            const int color_idx = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  55.3M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  436|  55.3M|                                      color_map_cdf[ctx[m]], pal_sz - 1);
  437|  55.3M|            pal_tmp[(i - j) * stride + j] = order[m][color_idx];
  438|  55.3M|        }
  439|  5.48M|    }
  440|       |
  441|   199k|    t->c->pal_dsp.pal_idx_finish(pal_idx, pal_tmp, bw4 * 4, bh4 * 4,
  442|   199k|                                 w4 * 4, h4 * 4);
  443|   199k|}
decode.c:order_palette:
  356|  5.49M|{
  357|  5.49M|    int have_top = i > first;
  358|       |
  359|  5.49M|    assert(pal_idx);
  ------------------
  |  Branch (359:5): [True: 5.49M, False: 618]
  ------------------
  360|  5.49M|    pal_idx += first + (i - first) * stride;
  361|  57.8M|    for (int j = first, n = 0; j >= last; have_top = 1, j--, n++, pal_idx += stride - 1) {
  ------------------
  |  Branch (361:32): [True: 52.5M, False: 5.25M]
  ------------------
  362|  52.5M|        const int have_left = j > 0;
  363|       |
  364|  52.5M|        assert(have_left || have_top);
  ------------------
  |  Branch (364:9): [True: 50.1M, False: 2.38M]
  |  Branch (364:9): [True: 2.38M, False: 0]
  ------------------
  365|       |
  366|  52.5M|#define add(v_in) do { \
  367|  52.5M|        const int v = v_in; \
  368|  52.5M|        assert((unsigned)v < 8U); \
  369|  52.5M|        order[n][o_idx++] = v; \
  370|  52.5M|        mask |= 1 << v; \
  371|  52.5M|    } while (0)
  372|       |
  373|  52.5M|        unsigned mask = 0;
  374|  52.5M|        int o_idx = 0;
  375|  52.5M|        if (!have_left) {
  ------------------
  |  Branch (375:13): [True: 2.38M, False: 50.1M]
  ------------------
  376|  2.38M|            ctx[n] = 0;
  377|  2.38M|            add(pal_idx[-stride]);
  ------------------
  |  |  366|  2.38M|#define add(v_in) do { \
  |  |  367|  2.38M|        const int v = v_in; \
  |  |  368|  2.38M|        assert((unsigned)v < 8U); \
  |  |  369|  2.38M|        order[n][o_idx++] = v; \
  |  |  370|  2.38M|        mask |= 1 << v; \
  |  |  371|  2.38M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 2.38M]
  |  |  ------------------
  ------------------
  |  Branch (377:13): [True: 2.38M, False: 18.4E]
  ------------------
  378|  50.1M|        } else if (!have_top) {
  ------------------
  |  Branch (378:20): [True: 3.11M, False: 47.0M]
  ------------------
  379|  3.11M|            ctx[n] = 0;
  380|  3.11M|            add(pal_idx[-1]);
  ------------------
  |  |  366|  3.11M|#define add(v_in) do { \
  |  |  367|  3.11M|        const int v = v_in; \
  |  |  368|  3.11M|        assert((unsigned)v < 8U); \
  |  |  369|  3.11M|        order[n][o_idx++] = v; \
  |  |  370|  3.11M|        mask |= 1 << v; \
  |  |  371|  3.11M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 3.11M]
  |  |  ------------------
  ------------------
  |  Branch (380:13): [True: 3.11M, False: 18.4E]
  ------------------
  381|  47.0M|        } else {
  382|  47.0M|            const int l = pal_idx[-1], t = pal_idx[-stride], tl = pal_idx[-(stride + 1)];
  383|  47.0M|            const int same_t_l = t == l;
  384|  47.0M|            const int same_t_tl = t == tl;
  385|  47.0M|            const int same_l_tl = l == tl;
  386|  47.0M|            const int same_all = same_t_l & same_t_tl & same_l_tl;
  387|       |
  388|  47.0M|            if (same_all) {
  ------------------
  |  Branch (388:17): [True: 21.2M, False: 25.8M]
  ------------------
  389|  21.2M|                ctx[n] = 4;
  390|  21.2M|                add(t);
  ------------------
  |  |  366|  21.2M|#define add(v_in) do { \
  |  |  367|  21.2M|        const int v = v_in; \
  |  |  368|  21.2M|        assert((unsigned)v < 8U); \
  |  |  369|  21.2M|        order[n][o_idx++] = v; \
  |  |  370|  21.2M|        mask |= 1 << v; \
  |  |  371|  21.2M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 21.2M]
  |  |  ------------------
  ------------------
  |  Branch (390:17): [True: 21.2M, False: 11.1k]
  ------------------
  391|  25.8M|            } else if (same_t_l) {
  ------------------
  |  Branch (391:24): [True: 4.90M, False: 20.9M]
  ------------------
  392|  4.90M|                ctx[n] = 3;
  393|  4.90M|                add(t);
  ------------------
  |  |  366|  4.90M|#define add(v_in) do { \
  |  |  367|  4.90M|        const int v = v_in; \
  |  |  368|  4.90M|        assert((unsigned)v < 8U); \
  |  |  369|  4.90M|        order[n][o_idx++] = v; \
  |  |  370|  4.90M|        mask |= 1 << v; \
  |  |  371|  4.90M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 4.90M]
  |  |  ------------------
  ------------------
  |  Branch (393:17): [True: 4.90M, False: 448]
  ------------------
  394|  4.90M|                add(tl);
  ------------------
  |  |  366|  4.90M|#define add(v_in) do { \
  |  |  367|  4.90M|        const int v = v_in; \
  |  |  368|  4.90M|        assert((unsigned)v < 8U); \
  |  |  369|  4.90M|        order[n][o_idx++] = v; \
  |  |  370|  4.90M|        mask |= 1 << v; \
  |  |  371|  4.90M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 4.90M]
  |  |  ------------------
  ------------------
  |  Branch (394:17): [True: 4.90M, False: 18.4E]
  ------------------
  395|  20.9M|            } else if (same_t_tl | same_l_tl) {
  ------------------
  |  Branch (395:24): [True: 16.9M, False: 3.92M]
  ------------------
  396|  16.9M|                ctx[n] = 2;
  397|  16.9M|                add(tl);
  ------------------
  |  |  366|  16.9M|#define add(v_in) do { \
  |  |  367|  16.9M|        const int v = v_in; \
  |  |  368|  16.9M|        assert((unsigned)v < 8U); \
  |  |  369|  16.9M|        order[n][o_idx++] = v; \
  |  |  370|  16.9M|        mask |= 1 << v; \
  |  |  371|  16.9M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 16.9M]
  |  |  ------------------
  ------------------
  |  Branch (397:17): [True: 16.9M, False: 4.63k]
  ------------------
  398|  16.9M|                add(same_t_tl ? l : t);
  ------------------
  |  |  366|  16.9M|#define add(v_in) do { \
  |  |  367|  33.9M|        const int v = v_in; \
  |  |  ------------------
  |  |  |  Branch (367:23): [True: 8.48M, False: 8.51M]
  |  |  ------------------
  |  |  368|  16.9M|        assert((unsigned)v < 8U); \
  |  |  369|  16.9M|        order[n][o_idx++] = v; \
  |  |  370|  16.9M|        mask |= 1 << v; \
  |  |  371|  16.9M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 16.9M]
  |  |  ------------------
  ------------------
  |  Branch (398:17): [True: 16.9M, False: 18.4E]
  ------------------
  399|  16.9M|            } else {
  400|  3.92M|                ctx[n] = 1;
  401|  3.92M|                add(imin(t, l));
  ------------------
  |  |  366|  3.92M|#define add(v_in) do { \
  |  |  367|  3.92M|        const int v = v_in; \
  |  |  368|  3.92M|        assert((unsigned)v < 8U); \
  |  |  369|  6.94M|        order[n][o_idx++] = v; \
  |  |  370|  6.94M|        mask |= 1 << v; \
  |  |  371|  6.94M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 6.94M]
  |  |  ------------------
  ------------------
  |  Branch (401:17): [True: 6.94M, False: 18.4E]
  ------------------
  402|  6.94M|                add(imax(t, l));
  ------------------
  |  |  366|  6.94M|#define add(v_in) do { \
  |  |  367|  6.94M|        const int v = v_in; \
  |  |  368|  6.94M|        assert((unsigned)v < 8U); \
  |  |  369|  6.94M|        order[n][o_idx++] = v; \
  |  |  370|  6.94M|        mask |= 1 << v; \
  |  |  371|  6.94M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 6.94M]
  |  |  ------------------
  ------------------
  |  Branch (402:17): [True: 6.94M, False: 2.68k]
  ------------------
  403|  6.94M|                add(tl);
  ------------------
  |  |  366|  6.94M|#define add(v_in) do { \
  |  |  367|  6.94M|        const int v = v_in; \
  |  |  368|  6.94M|        assert((unsigned)v < 8U); \
  |  |  369|  6.94M|        order[n][o_idx++] = v; \
  |  |  370|  6.94M|        mask |= 1 << v; \
  |  |  371|  6.94M|    } while (0)
  |  |  ------------------
  |  |  |  Branch (371:14): [Folded, False: 6.94M]
  |  |  ------------------
  ------------------
  |  Branch (403:17): [True: 6.94M, False: 4.20k]
  ------------------
  404|  6.94M|            }
  405|  47.0M|        }
  406|   461M|        for (unsigned m = 1, bit = 0; m < 0x100; m <<= 1, bit++)
  ------------------
  |  Branch (406:39): [True: 405M, False: 55.5M]
  ------------------
  407|   405M|            if (!(mask & m))
  ------------------
  |  Branch (407:17): [True: 323M, False: 82.8M]
  ------------------
  408|   323M|                order[n][o_idx++] = bit;
  409|       |        assert(o_idx == 8);
  ------------------
  |  Branch (409:9): [True: 52.3M, False: 3.23M]
  ------------------
  410|  55.5M|#undef add
  411|  55.5M|    }
  412|  5.49M|}
decode.c:splat_intraref:
  566|  5.05M|{
  567|  5.05M|    const refmvs_block ALIGN(tmpl, 16) = (refmvs_block) {
  568|  5.05M|        .ref.ref = { 0, -1 },
  569|  5.05M|        .mv.mv[0].n = INVALID_MV,
  ------------------
  |  |   40|  5.05M|#define INVALID_MV 0x80008000
  ------------------
  570|  5.05M|        .bs = bs,
  571|  5.05M|        .mf = 0,
  572|  5.05M|    };
  573|  5.05M|    c->refmvs_dsp.splat_mv(&t->rt.r[(t->by & 31) + 5], &tmpl, t->bx, bw4, bh4);
  574|  5.05M|}
decode.c:read_mv_residual:
  109|  2.11M|{
  110|  2.11M|    MsacContext *const msac = &ts->msac;
  111|  2.11M|    const enum MVJoint mv_joint =
  112|  2.11M|        dav1d_msac_decode_symbol_adapt4(msac, ts->cdf.mv.joint, N_MV_JOINTS - 1);
  ------------------
  |  |   47|  2.11M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  113|  2.11M|    if (mv_joint & MV_JOINT_V)
  ------------------
  |  Branch (113:9): [True: 1.61M, False: 503k]
  ------------------
  114|  1.61M|        ref_mv->y += read_mv_component_diff(msac, &ts->cdf.mv.comp[0], mv_prec);
  115|  2.11M|    if (mv_joint & MV_JOINT_H)
  ------------------
  |  Branch (115:9): [True: 1.49M, False: 625k]
  ------------------
  116|  1.49M|        ref_mv->x += read_mv_component_diff(msac, &ts->cdf.mv.comp[1], mv_prec);
  117|  2.11M|}
decode.c:read_mv_component_diff:
   79|  3.10M|{
   80|  3.10M|    const int sign = dav1d_msac_decode_bool_adapt(msac, mv_comp->sign);
  ------------------
  |  |   52|  3.10M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   81|  3.10M|    const int cl = dav1d_msac_decode_symbol_adapt16(msac, mv_comp->classes, 10);
  ------------------
  |  |   57|  3.10M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
   82|  3.10M|    int up, fp = 3, hp = 1;
   83|       |
   84|  3.10M|    if (!cl) {
  ------------------
  |  Branch (84:9): [True: 1.45M, False: 1.64M]
  ------------------
   85|  1.45M|        up = dav1d_msac_decode_bool_adapt(msac, mv_comp->class0);
  ------------------
  |  |   52|  1.45M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   86|  1.45M|        if (mv_prec >= 0) {  // !force_integer_mv
  ------------------
  |  Branch (86:13): [True: 740k, False: 710k]
  ------------------
   87|   740k|            fp = dav1d_msac_decode_symbol_adapt4(msac, mv_comp->class0_fp[up], 3);
  ------------------
  |  |   47|   740k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
   88|   740k|            if (mv_prec > 0) // allow_high_precision_mv
  ------------------
  |  Branch (88:17): [True: 354k, False: 386k]
  ------------------
   89|   354k|                hp = dav1d_msac_decode_bool_adapt(msac, mv_comp->class0_hp);
  ------------------
  |  |   52|   354k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   90|   740k|        }
   91|  1.64M|    } else {
   92|  1.64M|        up = 1 << cl;
   93|  16.3M|        for (int n = 0; n < cl; n++)
  ------------------
  |  Branch (93:25): [True: 14.6M, False: 1.64M]
  ------------------
   94|  14.6M|            up |= dav1d_msac_decode_bool_adapt(msac, mv_comp->classN[n]) << n;
  ------------------
  |  |   52|  14.6M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   95|  1.64M|        if (mv_prec >= 0) {  // !force_integer_mv
  ------------------
  |  Branch (95:13): [True: 263k, False: 1.38M]
  ------------------
   96|   263k|            fp = dav1d_msac_decode_symbol_adapt4(msac, mv_comp->classN_fp, 3);
  ------------------
  |  |   47|   263k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
   97|   263k|            if (mv_prec > 0) // allow_high_precision_mv
  ------------------
  |  Branch (97:17): [True: 112k, False: 151k]
  ------------------
   98|   112k|                hp = dav1d_msac_decode_bool_adapt(msac, mv_comp->classN_hp);
  ------------------
  |  |   52|   112k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
   99|   263k|        }
  100|  1.64M|    }
  101|       |
  102|  3.10M|    const int diff = ((up << 3) | (fp << 1) | hp) + 1;
  103|       |
  104|  3.10M|    return sign ? -diff : diff;
  ------------------
  |  Branch (104:12): [True: 2.29M, False: 806k]
  ------------------
  105|  3.10M|}
decode.c:read_vartx_tree:
  448|  5.26M|{
  449|  5.26M|    const Dav1dFrameContext *const f = t->f;
  450|  5.26M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  451|  5.26M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
  452|       |
  453|       |    // var-tx tree coding
  454|  5.26M|    uint16_t tx_split[2] = { 0 };
  455|  5.26M|    b->max_ytx = dav1d_max_txfm_size_for_bs[bs][0];
  456|  5.26M|    if (!b->skip && (f->frame_hdr->segmentation.lossless[b->seg_id] ||
  ------------------
  |  Branch (456:9): [True: 1.85M, False: 3.41M]
  |  Branch (456:22): [True: 10.8k, False: 1.84M]
  ------------------
  457|  1.84M|                     b->max_ytx == TX_4X4))
  ------------------
  |  Branch (457:22): [True: 102k, False: 1.73M]
  ------------------
  458|   114k|    {
  459|   114k|        b->max_ytx = b->uvtx = TX_4X4;
  460|   114k|        if (f->frame_hdr->txfm_mode == DAV1D_TX_SWITCHABLE) {
  ------------------
  |  Branch (460:13): [True: 32.0k, False: 82.2k]
  ------------------
  461|  32.0k|            dav1d_memset_pow2[b_dim[2]](&t->a->tx[bx4], TX_4X4);
  462|  32.0k|            dav1d_memset_pow2[b_dim[3]](&t->l.tx[by4], TX_4X4);
  463|  32.0k|        }
  464|  5.15M|    } else if (f->frame_hdr->txfm_mode != DAV1D_TX_SWITCHABLE || b->skip) {
  ------------------
  |  Branch (464:16): [True: 4.13M, False: 1.01M]
  |  Branch (464:66): [True: 509k, False: 505k]
  ------------------
  465|  4.64M|        if (f->frame_hdr->txfm_mode == DAV1D_TX_SWITCHABLE) {
  ------------------
  |  Branch (465:13): [True: 509k, False: 4.13M]
  ------------------
  466|   509k|            dav1d_memset_pow2[b_dim[2]](&t->a->tx[bx4], b_dim[2 + 0]);
  467|   509k|            dav1d_memset_pow2[b_dim[3]](&t->l.tx[by4], b_dim[2 + 1]);
  468|   509k|        }
  469|  4.64M|        b->uvtx = dav1d_max_txfm_size_for_bs[bs][f->cur.p.layout];
  470|  4.64M|    } else {
  471|   503k|        assert(bw4 <= 16 || bh4 <= 16 || b->max_ytx == TX_64X64);
  ------------------
  |  Branch (471:9): [True: 497k, False: 6.12k]
  |  Branch (471:9): [True: 2.60k, False: 3.52k]
  |  Branch (471:9): [True: 3.52k, False: 0]
  ------------------
  472|   505k|        int y, x, y_off, x_off;
  473|   505k|        const TxfmInfo *const ytx = &dav1d_txfm_dimensions[b->max_ytx];
  474|  1.01M|        for (y = 0, y_off = 0; y < bh4; y += ytx->h, y_off++) {
  ------------------
  |  Branch (474:32): [True: 510k, False: 505k]
  ------------------
  475|  1.03M|            for (x = 0, x_off = 0; x < bw4; x += ytx->w, x_off++) {
  ------------------
  |  Branch (475:36): [True: 520k, False: 510k]
  ------------------
  476|   520k|                read_tx_tree(t, b->max_ytx, 0, tx_split, x_off, y_off);
  477|       |                // contexts are updated inside read_tx_tree()
  478|   520k|                t->bx += ytx->w;
  479|   520k|            }
  480|   510k|            t->bx -= x;
  481|   510k|            t->by += ytx->h;
  482|   510k|        }
  483|   505k|        t->by -= y;
  484|   505k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   505k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 505k]
  |  |  ------------------
  |  |   35|   505k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   505k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  485|      0|            printf("Post-vartxtree[%x/%x]: r=%d\n",
  486|      0|                   tx_split[0], tx_split[1], t->ts->msac.rng);
  487|   505k|        b->uvtx = dav1d_max_txfm_size_for_bs[bs][f->cur.p.layout];
  488|   505k|    }
  489|  5.26M|    assert(!(tx_split[0] & ~0x33));
  ------------------
  |  Branch (489:5): [True: 5.26M, False: 178]
  ------------------
  490|  5.26M|    b->tx_split0 = (uint8_t)tx_split[0];
  491|  5.26M|    b->tx_split1 = tx_split[1];
  492|  5.26M|}
decode.c:read_tx_tree:
  123|   968k|{
  124|   968k|    const Dav1dFrameContext *const f = t->f;
  125|   968k|    const int bx4 = t->bx & 31, by4 = t->by & 31;
  126|   968k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[from];
  127|   968k|    const int txw = t_dim->lw, txh = t_dim->lh;
  128|   968k|    int is_split;
  129|       |
  130|   968k|    if (depth < 2 && from > (int) TX_4X4) {
  ------------------
  |  Branch (130:9): [True: 809k, False: 158k]
  |  Branch (130:22): [True: 809k, False: 1]
  ------------------
  131|   809k|        const int cat = 2 * (TX_64X64 - t_dim->max) - depth;
  132|   809k|        const int a = t->a->tx[bx4] < txw;
  133|   809k|        const int l = t->l.tx[by4] < txh;
  134|       |
  135|   809k|        is_split = dav1d_msac_decode_bool_adapt(&t->ts->msac,
  ------------------
  |  |   52|   809k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  136|   809k|                       t->ts->cdf.m.txpart[cat][a + l]);
  137|   809k|        if (is_split)
  ------------------
  |  Branch (137:13): [True: 216k, False: 593k]
  ------------------
  138|   216k|            masks[depth] |= 1 << (y_off * 4 + x_off);
  139|   809k|    } else {
  140|   158k|        is_split = 0;
  141|   158k|    }
  142|       |
  143|   968k|    if (is_split && t_dim->max > TX_8X8) {
  ------------------
  |  Branch (143:9): [True: 216k, False: 752k]
  |  Branch (143:21): [True: 155k, False: 60.7k]
  ------------------
  144|   155k|        const enum RectTxfmSize sub = t_dim->sub;
  145|   155k|        const TxfmInfo *const sub_t_dim = &dav1d_txfm_dimensions[sub];
  146|   155k|        const int txsw = sub_t_dim->w, txsh = sub_t_dim->h;
  147|       |
  148|   155k|        read_tx_tree(t, sub, depth + 1, masks, x_off * 2 + 0, y_off * 2 + 0);
  149|   155k|        t->bx += txsw;
  150|   155k|        if (txw >= txh && t->bx < f->bw)
  ------------------
  |  Branch (150:13): [True: 118k, False: 36.5k]
  |  Branch (150:27): [True: 117k, False: 769]
  ------------------
  151|   117k|            read_tx_tree(t, sub, depth + 1, masks, x_off * 2 + 1, y_off * 2 + 0);
  152|   155k|        t->bx -= txsw;
  153|   155k|        t->by += txsh;
  154|   155k|        if (txh >= txw && t->by < f->bh) {
  ------------------
  |  Branch (154:13): [True: 107k, False: 47.6k]
  |  Branch (154:27): [True: 106k, False: 1.30k]
  ------------------
  155|   106k|            read_tx_tree(t, sub, depth + 1, masks, x_off * 2 + 0, y_off * 2 + 1);
  156|   106k|            t->bx += txsw;
  157|   106k|            if (txw >= txh && t->bx < f->bw)
  ------------------
  |  Branch (157:17): [True: 69.7k, False: 36.5k]
  |  Branch (157:31): [True: 68.9k, False: 742]
  ------------------
  158|  68.9k|                read_tx_tree(t, sub, depth + 1, masks,
  159|  68.9k|                             x_off * 2 + 1, y_off * 2 + 1);
  160|   106k|            t->bx -= txsw;
  161|   106k|        }
  162|   155k|        t->by -= txsh;
  163|   812k|    } else {
  164|   812k|        dav1d_memset_pow2[t_dim->lw](&t->a->tx[bx4], is_split ? TX_4X4 : txw);
  ------------------
  |  Branch (164:54): [True: 60.8k, False: 752k]
  ------------------
  165|   812k|        dav1d_memset_pow2[t_dim->lh](&t->l.tx[by4], is_split ? TX_4X4 : txh);
  ------------------
  |  Branch (165:53): [True: 60.7k, False: 752k]
  ------------------
  166|   812k|    }
  167|   968k|}
decode.c:splat_intrabc_mv:
  535|  1.32M|{
  536|  1.32M|    const refmvs_block ALIGN(tmpl, 16) = (refmvs_block) {
  537|  1.32M|        .ref.ref = { 0, -1 },
  538|  1.32M|        .mv.mv[0] = b->mv[0],
  539|  1.32M|        .bs = bs,
  540|  1.32M|        .mf = 0,
  541|  1.32M|    };
  542|  1.32M|    c->refmvs_dsp.splat_mv(&t->rt.r[(t->by & 31) + 5], &tmpl, t->bx, bw4, bh4);
  543|  1.32M|}
decode.c:findoddzero:
  339|  1.89M|static inline int findoddzero(const uint8_t *buf, int len) {
  340|  2.13M|    for (int n = 0; n < len; n++)
  ------------------
  |  Branch (340:21): [True: 2.05M, False: 83.7k]
  ------------------
  341|  2.05M|        if (!buf[n * 2]) return 1;
  ------------------
  |  Branch (341:13): [True: 1.80M, False: 247k]
  ------------------
  342|  83.7k|    return 0;
  343|  1.89M|}
decode.c:find_matching_ref:
  197|  1.80M|{
  198|  1.80M|    /*const*/ refmvs_block *const *r = &t->rt.r[(t->by & 31) + 5];
  199|  1.80M|    int count = 0;
  200|  1.80M|    int have_topleft = have_top && have_left;
  ------------------
  |  Branch (200:24): [True: 1.69M, False: 114k]
  |  Branch (200:36): [True: 961k, False: 732k]
  ------------------
  201|  1.80M|    int have_topright = imax(bw4, bh4) < 32 &&
  ------------------
  |  Branch (201:25): [True: 1.24M, False: 559k]
  ------------------
  202|  1.24M|                        have_top && t->bx + bw4 < t->ts->tiling.col_end &&
  ------------------
  |  Branch (202:25): [True: 1.16M, False: 85.4k]
  |  Branch (202:37): [True: 917k, False: 246k]
  ------------------
  203|   917k|                        (intra_edge_flags & EDGE_I444_TOP_HAS_RIGHT);
  ------------------
  |  Branch (203:25): [True: 620k, False: 296k]
  ------------------
  204|       |
  205|  1.80M|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  206|  1.80M|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  207|       |
  208|  1.80M|    if (have_top) {
  ------------------
  |  Branch (208:9): [True: 1.69M, False: 114k]
  ------------------
  209|  1.69M|        const refmvs_block *r2 = &r[-1][t->bx];
  210|  1.69M|        if (matches(r2)) {
  ------------------
  |  |  206|  1.69M|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 1.52M, False: 167k]
  |  |  |  Branch (206:53): [True: 1.45M, False: 71.6k]
  |  |  ------------------
  ------------------
  211|  1.45M|            masks[0] |= 1;
  212|  1.45M|            count = 1;
  213|  1.45M|        }
  214|  1.69M|        int aw4 = bs(r2)[0];
  ------------------
  |  |  205|  1.69M|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  215|  1.69M|        if (aw4 >= bw4) {
  ------------------
  |  Branch (215:13): [True: 1.57M, False: 118k]
  ------------------
  216|  1.57M|            const int off = t->bx & (aw4 - 1);
  217|  1.57M|            if (off) have_topleft = 0;
  ------------------
  |  Branch (217:17): [True: 126k, False: 1.44M]
  ------------------
  218|  1.57M|            if (aw4 - off > bw4) have_topright = 0;
  ------------------
  |  Branch (218:17): [True: 131k, False: 1.44M]
  ------------------
  219|  1.57M|        } else {
  220|   118k|            unsigned mask = 1 << aw4;
  221|   290k|            for (int x = aw4; x < w4; x += aw4) {
  ------------------
  |  Branch (221:31): [True: 172k, False: 118k]
  ------------------
  222|   172k|                r2 += aw4;
  223|   172k|                if (matches(r2)) {
  ------------------
  |  |  206|   172k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 117k, False: 54.5k]
  |  |  |  Branch (206:53): [True: 107k, False: 9.62k]
  |  |  ------------------
  ------------------
  224|   107k|                    masks[0] |= mask;
  225|   107k|                    if (++count >= 8) return;
  ------------------
  |  Branch (225:25): [True: 453, False: 107k]
  ------------------
  226|   107k|                }
  227|   171k|                aw4 = bs(r2)[0];
  ------------------
  |  |  205|   171k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  228|   171k|                mask <<= aw4;
  229|   171k|            }
  230|   118k|        }
  231|  1.69M|    }
  232|  1.80M|    if (have_left) {
  ------------------
  |  Branch (232:9): [True: 1.07M, False: 732k]
  ------------------
  233|  1.07M|        /*const*/ refmvs_block *const *r2 = r;
  234|  1.07M|        if (matches(&r2[0][t->bx - 1])) {
  ------------------
  |  |  206|  1.07M|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 902k, False: 173k]
  |  |  |  Branch (206:53): [True: 828k, False: 73.9k]
  |  |  ------------------
  ------------------
  235|   828k|            masks[1] |= 1;
  236|   828k|            if (++count >= 8) return;
  ------------------
  |  Branch (236:17): [True: 241, False: 828k]
  ------------------
  237|   828k|        }
  238|  1.07M|        int lh4 = bs(&r2[0][t->bx - 1])[1];
  ------------------
  |  |  205|  1.07M|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  239|  1.07M|        if (lh4 >= bh4) {
  ------------------
  |  Branch (239:13): [True: 936k, False: 138k]
  ------------------
  240|   936k|            if (t->by & (lh4 - 1)) have_topleft = 0;
  ------------------
  |  Branch (240:17): [True: 156k, False: 780k]
  ------------------
  241|   936k|        } else {
  242|   138k|            unsigned mask = 1 << lh4;
  243|   334k|            for (int y = lh4; y < h4; y += lh4) {
  ------------------
  |  Branch (243:31): [True: 197k, False: 137k]
  ------------------
  244|   197k|                r2 += lh4;
  245|   197k|                if (matches(&r2[0][t->bx - 1])) {
  ------------------
  |  |  206|   197k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 136k, False: 60.9k]
  |  |  |  Branch (206:53): [True: 125k, False: 10.6k]
  |  |  ------------------
  ------------------
  246|   125k|                    masks[1] |= mask;
  247|   125k|                    if (++count >= 8) return;
  ------------------
  |  Branch (247:25): [True: 1.41k, False: 124k]
  ------------------
  248|   125k|                }
  249|   195k|                lh4 = bs(&r2[0][t->bx - 1])[1];
  ------------------
  |  |  205|   195k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  250|   195k|                mask <<= lh4;
  251|   195k|            }
  252|   138k|        }
  253|  1.07M|    }
  254|  1.80M|    if (have_topleft && matches(&r[-1][t->bx - 1])) {
  ------------------
  |  |  206|   676k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 544k, False: 131k]
  |  |  |  Branch (206:53): [True: 500k, False: 44.2k]
  |  |  ------------------
  ------------------
  |  Branch (254:9): [True: 676k, False: 1.12M]
  ------------------
  255|   500k|        masks[1] |= 1ULL << 32;
  256|   500k|        if (++count >= 8) return;
  ------------------
  |  Branch (256:13): [True: 944, False: 499k]
  ------------------
  257|   500k|    }
  258|  1.80M|    if (have_topright && matches(&r[-1][t->bx + bw4])) {
  ------------------
  |  |  206|   489k|#define matches(rp) ((rp)->ref.ref[0] == ref + 1 && (rp)->ref.ref[1] == -1)
  |  |  ------------------
  |  |  |  Branch (206:22): [True: 398k, False: 91.6k]
  |  |  |  Branch (206:53): [True: 371k, False: 26.9k]
  |  |  ------------------
  ------------------
  |  Branch (258:9): [True: 489k, False: 1.31M]
  ------------------
  259|   371k|        masks[0] |= 1ULL << 32;
  260|   371k|    }
  261|  1.80M|#undef matches
  262|  1.80M|}
decode.c:derive_warpmv:
  268|   169k|{
  269|   169k|    int pts[8][2 /* in, out */][2 /* x, y */], np = 0;
  270|   169k|    /*const*/ refmvs_block *const *r = &t->rt.r[(t->by & 31) + 5];
  271|       |
  272|   169k|#define add_sample(dx, dy, sx, sy, rp) do { \
  273|   169k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  274|   169k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  275|   169k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  276|   169k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  277|   169k|    np++; \
  278|   169k|} while (0)
  279|       |
  280|       |    // use masks[] to find the projectable motion vectors in the edges
  281|   169k|    if ((unsigned) masks[0] == 1 && !(masks[1] >> 32)) {
  ------------------
  |  Branch (281:9): [True: 118k, False: 51.8k]
  |  Branch (281:37): [True: 60.6k, False: 57.4k]
  ------------------
  282|  60.6k|        const int off = t->bx & (bs(&r[-1][t->bx])[0] - 1);
  ------------------
  |  |  205|  60.6k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  283|  60.6k|        add_sample(-off, 0, 1, -1, &r[-1][t->bx]);
  ------------------
  |  |  272|  60.6k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  60.6k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  60.6k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  60.6k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  60.6k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  60.6k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  60.6k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  60.6k|    np++; \
  |  |  278|  60.6k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 60.6k]
  |  |  ------------------
  ------------------
  284|   203k|    } else for (unsigned off = 0, xmask = (uint32_t) masks[0]; np < 8 && xmask;) { // top
  ------------------
  |  Branch (284:64): [True: 203k, False: 379]
  |  Branch (284:74): [True: 94.5k, False: 108k]
  ------------------
  285|  94.5k|        const int tz = ctz(xmask);
  286|  94.5k|        off += tz;
  287|  94.5k|        xmask >>= tz;
  288|  94.5k|        add_sample(off, 0, 1, -1, &r[-1][t->bx + off]);
  ------------------
  |  |  272|  94.5k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  94.5k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  94.5k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  94.5k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  94.5k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  94.5k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  94.5k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  94.5k|    np++; \
  |  |  278|  94.5k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 94.5k]
  |  |  ------------------
  ------------------
  289|  94.5k|        xmask &= ~1;
  290|  94.5k|    }
  291|   169k|    if (np < 8 && masks[1] == 1) {
  ------------------
  |  Branch (291:9): [True: 169k, False: 362]
  |  Branch (291:19): [True: 59.9k, False: 109k]
  ------------------
  292|  59.9k|        const int off = t->by & (bs(&r[0][t->bx - 1])[1] - 1);
  ------------------
  |  |  205|  59.9k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  ------------------
  293|  59.9k|        add_sample(0, -off, -1, 1, &r[-off][t->bx - 1]);
  ------------------
  |  |  272|  59.9k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  59.9k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  59.9k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  59.9k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  59.9k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  59.9k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  59.9k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  59.9k|    np++; \
  |  |  278|  59.9k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 59.9k]
  |  |  ------------------
  ------------------
  294|   207k|    } else for (unsigned off = 0, ymask = (uint32_t) masks[1]; np < 8 && ymask;) { // left
  ------------------
  |  Branch (294:64): [True: 207k, False: 831]
  |  Branch (294:74): [True: 98.0k, False: 109k]
  ------------------
  295|  98.0k|        const int tz = ctz(ymask);
  296|  98.0k|        off += tz;
  297|  98.0k|        ymask >>= tz;
  298|  98.0k|        add_sample(0, off, -1, 1, &r[off][t->bx - 1]);
  ------------------
  |  |  272|  98.0k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  98.0k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  98.0k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  98.0k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  98.0k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  98.0k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  98.0k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  98.0k|    np++; \
  |  |  278|  98.0k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 98.0k]
  |  |  ------------------
  ------------------
  299|  98.0k|        ymask &= ~1;
  300|  98.0k|    }
  301|   169k|    if (np < 8 && masks[1] >> 32) // top/left
  ------------------
  |  Branch (301:9): [True: 168k, False: 974]
  |  Branch (301:19): [True: 75.6k, False: 93.2k]
  ------------------
  302|  75.6k|        add_sample(0, 0, -1, -1, &r[-1][t->bx - 1]);
  ------------------
  |  |  272|  75.6k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  75.6k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  75.6k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  75.6k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  75.6k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  75.6k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  75.6k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  75.6k|    np++; \
  |  |  278|  75.6k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 75.6k]
  |  |  ------------------
  ------------------
  303|   169k|    if (np < 8 && masks[0] >> 32) // top/right
  ------------------
  |  Branch (303:9): [True: 168k, False: 1.20k]
  |  Branch (303:19): [True: 46.6k, False: 122k]
  ------------------
  304|  46.6k|        add_sample(bw4, 0, 1, -1, &r[-1][t->bx + bw4]);
  ------------------
  |  |  272|  46.6k|#define add_sample(dx, dy, sx, sy, rp) do { \
  |  |  273|  46.6k|    pts[np][0][0] = 16 * (2 * dx + sx * bs(rp)[0]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  46.6k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  274|  46.6k|    pts[np][0][1] = 16 * (2 * dy + sy * bs(rp)[1]) - 8; \
  |  |  ------------------
  |  |  |  |  205|  46.6k|#define bs(rp) dav1d_block_dimensions[(rp)->bs]
  |  |  ------------------
  |  |  275|  46.6k|    pts[np][1][0] = pts[np][0][0] + (rp)->mv.mv[0].x; \
  |  |  276|  46.6k|    pts[np][1][1] = pts[np][0][1] + (rp)->mv.mv[0].y; \
  |  |  277|  46.6k|    np++; \
  |  |  278|  46.6k|} while (0)
  |  |  ------------------
  |  |  |  Branch (278:10): [Folded, False: 46.6k]
  |  |  ------------------
  ------------------
  305|   169k|    assert(np > 0 && np <= 8);
  ------------------
  |  Branch (305:5): [True: 169k, False: 204]
  |  Branch (305:5): [True: 169k, False: 0]
  ------------------
  306|   169k|#undef bs
  307|       |
  308|       |    // select according to motion vector difference against a threshold
  309|   169k|    int mvd[8], ret = 0;
  310|   169k|    const int thresh = 4 * iclip(imax(bw4, bh4), 4, 28);
  311|   604k|    for (int i = 0; i < np; i++) {
  ------------------
  |  Branch (311:21): [True: 435k, False: 169k]
  ------------------
  312|   435k|        mvd[i] = abs(pts[i][1][0] - pts[i][0][0] - mv.x) +
  313|   435k|                 abs(pts[i][1][1] - pts[i][0][1] - mv.y);
  314|   435k|        if (mvd[i] > thresh)
  ------------------
  |  Branch (314:13): [True: 145k, False: 289k]
  ------------------
  315|   145k|            mvd[i] = -1;
  316|   289k|        else
  317|   289k|            ret++;
  318|   435k|    }
  319|   169k|    if (!ret) {
  ------------------
  |  Branch (319:9): [True: 33.9k, False: 135k]
  ------------------
  320|  33.9k|        ret = 1;
  321|   158k|    } else for (int i = 0, j = np - 1, k = 0; k < np - ret; k++, i++, j--) {
  ------------------
  |  Branch (321:47): [True: 58.5k, False: 100k]
  ------------------
  322|   100k|        while (mvd[i] != -1) i++;
  ------------------
  |  Branch (322:16): [True: 41.7k, False: 58.5k]
  ------------------
  323|   109k|        while (mvd[j] == -1) j--;
  ------------------
  |  Branch (323:16): [True: 50.8k, False: 58.5k]
  ------------------
  324|  58.5k|        assert(i != j);
  ------------------
  |  Branch (324:9): [True: 58.5k, False: 0]
  ------------------
  325|  58.5k|        if (i > j) break;
  ------------------
  |  Branch (325:13): [True: 35.6k, False: 22.8k]
  ------------------
  326|       |        // replace the discarded samples;
  327|  22.8k|        mvd[i] = mvd[j];
  328|  22.8k|        memcpy(pts[i], pts[j], sizeof(*pts));
  329|  22.8k|    }
  330|       |
  331|   169k|    if (!dav1d_find_affine_int(pts, ret, bw4, bh4, mv, wmp, t->bx, t->by) &&
  ------------------
  |  Branch (331:9): [True: 161k, False: 8.30k]
  ------------------
  332|   161k|        !dav1d_get_shear_params(wmp))
  ------------------
  |  Branch (332:9): [True: 153k, False: 7.62k]
  ------------------
  333|   153k|    {
  334|   153k|        wmp->type = DAV1D_WM_TYPE_AFFINE;
  335|   153k|    } else
  336|  15.9k|        wmp->type = DAV1D_WM_TYPE_IDENTITY;
  337|   169k|}
decode.c:splat_tworef_mv:
  550|   398k|{
  551|   398k|    assert(bw4 >= 2 && bh4 >= 2);
  ------------------
  |  Branch (551:5): [True: 398k, False: 52]
  |  Branch (551:5): [True: 398k, False: 18.4E]
  ------------------
  552|   398k|    const enum CompInterPredMode mode = b->inter_mode;
  553|   398k|    const refmvs_block ALIGN(tmpl, 16) = (refmvs_block) {
  554|   398k|        .ref.ref = { b->ref[0] + 1, b->ref[1] + 1 },
  555|   398k|        .mv.mv = { b->mv[0], b->mv[1] },
  556|   398k|        .bs = bs,
  557|   398k|        .mf = (mode == GLOBALMV_GLOBALMV) | !!((1 << mode) & (0xbc)) * 2,
  558|   398k|    };
  559|   398k|    c->refmvs_dsp.splat_mv(&t->rt.r[(t->by & 31) + 5], &tmpl, t->bx, bw4, bh4);
  560|   398k|}
decode.c:splat_oneref_mv:
  519|  3.54M|{
  520|  3.54M|    const enum InterPredMode mode = b->inter_mode;
  521|  3.54M|    const refmvs_block ALIGN(tmpl, 16) = (refmvs_block) {
  522|  3.54M|        .ref.ref = { b->ref[0] + 1, b->interintra_type ? 0 : -1 },
  ------------------
  |  Branch (522:37): [True: 113k, False: 3.42M]
  ------------------
  523|  3.54M|        .mv.mv[0] = b->mv[0],
  524|  3.54M|        .bs = bs,
  525|  3.54M|        .mf = (mode == GLOBALMV && imin(bw4, bh4) >= 2) | ((mode == NEWMV) * 2),
  ------------------
  |  Branch (525:16): [True: 2.05M, False: 1.48M]
  |  Branch (525:36): [True: 1.92M, False: 129k]
  ------------------
  526|  3.54M|    };
  527|  3.54M|    c->refmvs_dsp.splat_mv(&t->rt.r[(t->by & 31) + 5], &tmpl, t->bx, bw4, bh4);
  528|  3.54M|}
decode.c:affine_lowest_px_luma:
  619|   936k|{
  620|   936k|    affine_lowest_px(t, dst, b_dim, wmp, 0, 0);
  621|   936k|}
decode.c:affine_lowest_px:
  597|  1.00M|{
  598|  1.00M|    const int h_mul = 4 >> ss_hor, v_mul = 4 >> ss_ver;
  599|  1.00M|    assert(!((b_dim[0] * h_mul) & 7) && !((b_dim[1] * v_mul) & 7));
  ------------------
  |  Branch (599:5): [True: 1.00M, False: 18.4E]
  |  Branch (599:5): [True: 1.00M, False: 6]
  ------------------
  600|  1.00M|    const int32_t *const mat = wmp->matrix;
  601|  1.00M|    const int y = b_dim[1] * v_mul - 8; // lowest y
  602|       |
  603|  1.00M|    const int src_y = t->by * 4 + ((y + 4) << ss_ver);
  604|  1.00M|    const int64_t mat5_y = (int64_t) mat[5] * src_y + mat[1];
  605|       |    // check left- and right-most blocks
  606|  2.93M|    for (int x = 0; x < b_dim[0] * h_mul; x += imax(8, b_dim[0] * h_mul - 8)) {
  ------------------
  |  Branch (606:21): [True: 1.92M, False: 1.00M]
  ------------------
  607|       |        // calculate transformation relative to center of 8x8 block in
  608|       |        // luma pixel units
  609|  1.92M|        const int src_x = t->bx * 4 + ((x + 4) << ss_hor);
  610|  1.92M|        const int64_t mvy = ((int64_t) mat[4] * src_x + mat5_y) >> ss_ver;
  611|  1.92M|        const int dy = (int) (mvy >> 16) - 4;
  612|  1.92M|        *dst = imax(*dst, dy + 4 + 8);
  613|  1.92M|    }
  614|  1.00M|}
decode.c:mc_lowest_px:
  579|  6.23M|{
  580|  6.23M|    const int v_mul = 4 >> ss_ver;
  581|  6.23M|    if (!smp->scale) {
  ------------------
  |  Branch (581:9): [True: 4.43M, False: 1.80M]
  ------------------
  582|  4.43M|        const int my = mvy >> (3 + ss_ver), dy = mvy & (15 >> !ss_ver);
  583|  4.43M|        *dst = imax(*dst, (by4 + bh4) * v_mul + my + 4 * !!dy);
  584|  4.43M|    } else {
  585|  1.80M|        int y = (by4 * v_mul << 4) + mvy * (1 << !ss_ver);
  586|  1.80M|        const int64_t tmp = (int64_t)(y) * smp->scale + (smp->scale - 0x4000) * 8;
  587|  1.80M|        y = apply_sign64((int)((llabs(tmp) + 128) >> 8), tmp) + 32;
  588|  1.80M|        const int bottom = ((y + (bh4 * v_mul - 1) * smp->step) >> 10) + 1 + 4;
  589|  1.80M|        *dst = imax(*dst, bottom);
  590|  1.80M|    }
  591|  6.23M|}
decode.c:obmc_lowest_px:
  639|   731k|{
  640|   731k|    assert(!(t->bx & 1) && !(t->by & 1));
  ------------------
  |  Branch (640:5): [True: 731k, False: 18]
  |  Branch (640:5): [True: 731k, False: 18.4E]
  ------------------
  641|   731k|    const Dav1dFrameContext *const f = t->f;
  642|   731k|    /*const*/ refmvs_block **r = &t->rt.r[(t->by & 31) + 5];
  643|   731k|    const int ss_ver = is_chroma && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (643:24): [True: 310k, False: 421k]
  |  Branch (643:37): [True: 150k, False: 159k]
  ------------------
  644|   731k|    const int ss_hor = is_chroma && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (644:24): [True: 310k, False: 421k]
  |  Branch (644:37): [True: 150k, False: 159k]
  ------------------
  645|   731k|    const int h_mul = 4 >> ss_hor, v_mul = 4 >> ss_ver;
  646|       |
  647|   731k|    if (t->by > t->ts->tiling.row_start &&
  ------------------
  |  Branch (647:9): [True: 624k, False: 106k]
  ------------------
  648|   624k|        (!is_chroma || b_dim[0] * h_mul + b_dim[1] * v_mul >= 16))
  ------------------
  |  Branch (648:10): [True: 361k, False: 263k]
  |  Branch (648:24): [True: 158k, False: 104k]
  ------------------
  649|   520k|    {
  650|  1.09M|        for (int i = 0, x = 0; x < w4 && i < imin(b_dim[2], 4); ) {
  ------------------
  |  Branch (650:32): [True: 578k, False: 518k]
  |  Branch (650:42): [True: 575k, False: 2.21k]
  ------------------
  651|       |            // only odd blocks are considered for overlap handling, hence +1
  652|   575k|            const refmvs_block *const a_r = &r[-1][t->bx + x + 1];
  653|   575k|            const uint8_t *const a_b_dim = dav1d_block_dimensions[a_r->bs];
  654|       |
  655|   575k|            if (a_r->ref.ref[0] > 0) {
  ------------------
  |  Branch (655:17): [True: 548k, False: 27.4k]
  ------------------
  656|   548k|                const int oh4 = imin(b_dim[1], 16) >> 1;
  657|   548k|                mc_lowest_px(&dst[a_r->ref.ref[0] - 1][is_chroma], t->by,
  658|   548k|                             (oh4 * 3 + 3) >> 2, a_r->mv.mv[0].y, ss_ver,
  659|   548k|                             &f->svc[a_r->ref.ref[0] - 1][1]);
  660|   548k|                i++;
  661|   548k|            }
  662|   575k|            x += imax(a_b_dim[0], 2);
  663|   575k|        }
  664|   520k|    }
  665|       |
  666|   731k|    if (t->bx > t->ts->tiling.col_start)
  ------------------
  |  Branch (666:9): [True: 710k, False: 20.6k]
  ------------------
  667|  1.49M|        for (int i = 0, y = 0; y < h4 && i < imin(b_dim[3], 4); ) {
  ------------------
  |  Branch (667:32): [True: 790k, False: 709k]
  |  Branch (667:42): [True: 788k, False: 1.58k]
  ------------------
  668|       |            // only odd blocks are considered for overlap handling, hence +1
  669|   788k|            const refmvs_block *const l_r = &r[y + 1][t->bx - 1];
  670|   788k|            const uint8_t *const l_b_dim = dav1d_block_dimensions[l_r->bs];
  671|       |
  672|   788k|            if (l_r->ref.ref[0] > 0) {
  ------------------
  |  Branch (672:17): [True: 749k, False: 39.2k]
  ------------------
  673|   749k|                const int oh4 = iclip(l_b_dim[1], 2, b_dim[1]);
  674|   749k|                mc_lowest_px(&dst[l_r->ref.ref[0] - 1][is_chroma],
  675|   749k|                             t->by + y, oh4, l_r->mv.mv[0].y, ss_ver,
  676|   749k|                             &f->svc[l_r->ref.ref[0] - 1][1]);
  677|   749k|                i++;
  678|   749k|            }
  679|   788k|            y += imax(l_b_dim[1], 2);
  680|   788k|        }
  681|   731k|}
decode.c:affine_lowest_px_chroma:
  626|  89.5k|{
  627|  89.5k|    const Dav1dFrameContext *const f = t->f;
  628|  89.5k|    assert(f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400);
  ------------------
  |  Branch (628:5): [True: 89.5k, False: 1]
  ------------------
  629|  89.5k|    if (f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I444)
  ------------------
  |  Branch (629:9): [True: 17.4k, False: 72.0k]
  ------------------
  630|  17.4k|        affine_lowest_px_luma(t, dst, b_dim, wmp);
  631|  72.0k|    else
  632|  72.0k|        affine_lowest_px(t, dst, b_dim, wmp, f->cur.p.layout & DAV1D_PIXEL_LAYOUT_I420, 1);
  633|  89.5k|}
decode.c:read_restoration_info:
 2516|   260k|{
 2517|   260k|    const Dav1dFrameContext *const f = t->f;
 2518|   260k|    Dav1dTileState *const ts = t->ts;
 2519|   260k|    const Av1RestorationUnit *const lr_ref = ts->lr_ref[p];
 2520|       |
 2521|   260k|    if (frame_type == DAV1D_RESTORATION_SWITCHABLE) {
  ------------------
  |  Branch (2521:9): [True: 135k, False: 125k]
  ------------------
 2522|   135k|        const int filter = dav1d_msac_decode_symbol_adapt4(&ts->msac,
  ------------------
  |  |   47|   135k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
 2523|   135k|                               ts->cdf.m.restore_switchable, 2);
 2524|   135k|        lr->type = filter + !!filter; /* NONE/WIENER/SGRPROJ */
 2525|   135k|    } else {
 2526|   125k|        const unsigned type =
 2527|   125k|            dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   125k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
 2528|   125k|                frame_type == DAV1D_RESTORATION_WIENER ?
  ------------------
  |  Branch (2528:17): [True: 41.3k, False: 83.7k]
  ------------------
 2529|  83.7k|                ts->cdf.m.restore_wiener : ts->cdf.m.restore_sgrproj);
 2530|   125k|        lr->type = type ? frame_type : DAV1D_RESTORATION_NONE;
  ------------------
  |  Branch (2530:20): [True: 53.3k, False: 71.7k]
  ------------------
 2531|   125k|    }
 2532|       |
 2533|   260k|    if (lr->type == DAV1D_RESTORATION_WIENER) {
  ------------------
  |  Branch (2533:9): [True: 33.1k, False: 227k]
  ------------------
 2534|  33.1k|        lr->filter_v[0] = p ? 0 :
  ------------------
  |  Branch (2534:27): [True: 19.3k, False: 13.7k]
  ------------------
 2535|  33.1k|            dav1d_msac_decode_subexp(&ts->msac,
 2536|  13.7k|                lr_ref->filter_v[0] + 5, 16, 1) - 5;
 2537|  33.1k|        lr->filter_v[1] =
 2538|  33.1k|            dav1d_msac_decode_subexp(&ts->msac,
 2539|  33.1k|                lr_ref->filter_v[1] + 23, 32, 2) - 23;
 2540|  33.1k|        lr->filter_v[2] =
 2541|  33.1k|            dav1d_msac_decode_subexp(&ts->msac,
 2542|  33.1k|                lr_ref->filter_v[2] + 17, 64, 3) - 17;
 2543|       |
 2544|  33.1k|        lr->filter_h[0] = p ? 0 :
  ------------------
  |  Branch (2544:27): [True: 19.3k, False: 13.7k]
  ------------------
 2545|  33.1k|            dav1d_msac_decode_subexp(&ts->msac,
 2546|  13.7k|                lr_ref->filter_h[0] + 5, 16, 1) - 5;
 2547|  33.1k|        lr->filter_h[1] =
 2548|  33.1k|            dav1d_msac_decode_subexp(&ts->msac,
 2549|  33.1k|                lr_ref->filter_h[1] + 23, 32, 2) - 23;
 2550|  33.1k|        lr->filter_h[2] =
 2551|  33.1k|            dav1d_msac_decode_subexp(&ts->msac,
 2552|  33.1k|                lr_ref->filter_h[2] + 17, 64, 3) - 17;
 2553|  33.1k|        memcpy(lr->sgr_weights, lr_ref->sgr_weights, sizeof(lr->sgr_weights));
 2554|  33.1k|        ts->lr_ref[p] = lr;
 2555|  33.1k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  33.1k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 33.1k]
  |  |  ------------------
  |  |   35|  33.1k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  33.1k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2556|      0|            printf("Post-lr_wiener[pl=%d,v[%d,%d,%d],h[%d,%d,%d]]: r=%d\n",
 2557|      0|                   p, lr->filter_v[0], lr->filter_v[1],
 2558|      0|                   lr->filter_v[2], lr->filter_h[0],
 2559|      0|                   lr->filter_h[1], lr->filter_h[2], ts->msac.rng);
 2560|   227k|    } else if (lr->type == DAV1D_RESTORATION_SGRPROJ) {
  ------------------
  |  Branch (2560:16): [True: 51.7k, False: 175k]
  ------------------
 2561|  51.7k|        const unsigned idx = dav1d_msac_decode_bools(&ts->msac, 4);
 2562|  51.7k|        const uint16_t *const sgr_params = dav1d_sgr_params[idx];
 2563|  51.7k|        lr->type += idx;
 2564|  51.7k|        lr->sgr_weights[0] = sgr_params[0] ? dav1d_msac_decode_subexp(&ts->msac,
  ------------------
  |  Branch (2564:30): [True: 42.5k, False: 9.16k]
  ------------------
 2565|  42.5k|            lr_ref->sgr_weights[0] + 96, 128, 4) - 96 : 0;
 2566|  51.7k|        lr->sgr_weights[1] = sgr_params[1] ? dav1d_msac_decode_subexp(&ts->msac,
  ------------------
  |  Branch (2566:30): [True: 29.0k, False: 22.7k]
  ------------------
 2567|  29.0k|            lr_ref->sgr_weights[1] + 32, 128, 4) - 32 : 95;
 2568|  51.7k|        memcpy(lr->filter_v, lr_ref->filter_v, sizeof(lr->filter_v));
 2569|  51.7k|        memcpy(lr->filter_h, lr_ref->filter_h, sizeof(lr->filter_h));
 2570|  51.7k|        ts->lr_ref[p] = lr;
 2571|  51.7k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  51.7k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 51.7k]
  |  |  ------------------
  |  |   35|  51.7k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  51.7k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2572|      0|            printf("Post-lr_sgrproj[pl=%d,idx=%d,w[%d,%d]]: r=%d\n",
 2573|      0|                   p, idx, lr->sgr_weights[0],
 2574|      0|                   lr->sgr_weights[1], ts->msac.rng);
 2575|  51.7k|    }
 2576|   260k|}
decode.c:init_quant_tables:
   57|   375k|{
   58|   941k|    for (int i = 0; i < (frame_hdr->segmentation.enabled ? 8 : 1); i++) {
  ------------------
  |  Branch (58:21): [True: 565k, False: 375k]
  |  Branch (58:26): [True: 244k, False: 697k]
  ------------------
   59|   565k|        const int yac = frame_hdr->segmentation.enabled ?
  ------------------
  |  Branch (59:25): [True: 217k, False: 348k]
  ------------------
   60|   348k|            iclip_u8(qidx + frame_hdr->segmentation.seg_data.d[i].delta_q) : qidx;
   61|   565k|        const int ydc = iclip_u8(yac + frame_hdr->quant.ydc_delta);
   62|   565k|        const int uac = iclip_u8(yac + frame_hdr->quant.uac_delta);
   63|   565k|        const int udc = iclip_u8(yac + frame_hdr->quant.udc_delta);
   64|   565k|        const int vac = iclip_u8(yac + frame_hdr->quant.vac_delta);
   65|   565k|        const int vdc = iclip_u8(yac + frame_hdr->quant.vdc_delta);
   66|       |
   67|   565k|        dq[i][0][0] = dav1d_dq_tbl[seq_hdr->hbd][ydc][0];
   68|   565k|        dq[i][0][1] = dav1d_dq_tbl[seq_hdr->hbd][yac][1];
   69|   565k|        dq[i][1][0] = dav1d_dq_tbl[seq_hdr->hbd][udc][0];
   70|   565k|        dq[i][1][1] = dav1d_dq_tbl[seq_hdr->hbd][uac][1];
   71|   565k|        dq[i][2][0] = dav1d_dq_tbl[seq_hdr->hbd][vdc][0];
   72|   565k|        dq[i][2][1] = dav1d_dq_tbl[seq_hdr->hbd][vac][1];
   73|   565k|    }
   74|   375k|}
decode.c:setup_tile:
 2432|   315k|{
 2433|   315k|    const int col_sb_start = f->frame_hdr->tiling.col_start_sb[tile_col];
 2434|   315k|    const int col_sb128_start = col_sb_start >> !f->seq_hdr->sb128;
 2435|   315k|    const int col_sb_end = f->frame_hdr->tiling.col_start_sb[tile_col + 1];
 2436|   315k|    const int row_sb_start = f->frame_hdr->tiling.row_start_sb[tile_row];
 2437|   315k|    const int row_sb_end = f->frame_hdr->tiling.row_start_sb[tile_row + 1];
 2438|   315k|    const int sb_shift = f->sb_shift;
 2439|       |
 2440|   315k|    const uint8_t *const size_mul = ss_size_mul[f->cur.p.layout];
 2441|   946k|    for (int p = 0; p < 2; p++) {
  ------------------
  |  Branch (2441:21): [True: 630k, False: 315k]
  ------------------
 2442|   630k|        ts->frame_thread[p].pal_idx = f->frame_thread.pal_idx ?
  ------------------
  |  Branch (2442:39): [True: 543k, False: 86.7k]
  ------------------
 2443|   543k|            &f->frame_thread.pal_idx[(size_t)tile_start_off * size_mul[1] / 8] :
 2444|   630k|            NULL;
 2445|   630k|        ts->frame_thread[p].cbi = f->frame_thread.cbi ?
  ------------------
  |  Branch (2445:35): [True: 630k, False: 62]
  ------------------
 2446|   630k|            &f->frame_thread.cbi[(size_t)tile_start_off * size_mul[0] / 64] :
 2447|   630k|            NULL;
 2448|   630k|        ts->frame_thread[p].cf = f->frame_thread.cf ?
  ------------------
  |  Branch (2448:34): [True: 630k, False: 66]
  ------------------
 2449|   630k|            (uint8_t*)f->frame_thread.cf +
 2450|   630k|                (((size_t)tile_start_off * size_mul[0]) >> !f->seq_hdr->hbd) :
 2451|   630k|            NULL;
 2452|   630k|    }
 2453|       |
 2454|   315k|    dav1d_cdf_thread_copy(&ts->cdf, &f->in_cdf);
 2455|   315k|    ts->last_qidx = f->frame_hdr->quant.yac;
 2456|   315k|    ts->last_delta_lf.u32 = 0;
 2457|       |
 2458|   315k|    dav1d_msac_init(&ts->msac, data, sz, f->frame_hdr->disable_cdf_update);
 2459|       |
 2460|   315k|    ts->tiling.row = tile_row;
 2461|   315k|    ts->tiling.col = tile_col;
 2462|   315k|    ts->tiling.col_start = col_sb_start << sb_shift;
 2463|   315k|    ts->tiling.col_end = imin(col_sb_end << sb_shift, f->bw);
 2464|   315k|    ts->tiling.row_start = row_sb_start << sb_shift;
 2465|   315k|    ts->tiling.row_end = imin(row_sb_end << sb_shift, f->bh);
 2466|       |
 2467|       |    // Reference Restoration Unit (used for exp coding)
 2468|   315k|    int sb_idx, unit_idx;
 2469|   315k|    if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (2469:9): [True: 20.3k, False: 295k]
  ------------------
 2470|       |        // vertical components only
 2471|  20.3k|        sb_idx = (ts->tiling.row_start >> 5) * f->sr_sb128w;
 2472|  20.3k|        unit_idx = (ts->tiling.row_start & 16) >> 3;
 2473|   295k|    } else {
 2474|   295k|        sb_idx = (ts->tiling.row_start >> 5) * f->sb128w + col_sb128_start;
 2475|   295k|        unit_idx = ((ts->tiling.row_start & 16) >> 3) +
 2476|   295k|                   ((ts->tiling.col_start & 16) >> 4);
 2477|   295k|    }
 2478|  1.26M|    for (int p = 0; p < 3; p++) {
  ------------------
  |  Branch (2478:21): [True: 945k, False: 315k]
  ------------------
 2479|   945k|        if (!((f->lf.restore_planes >> p) & 1U))
  ------------------
  |  Branch (2479:13): [True: 890k, False: 54.8k]
  ------------------
 2480|   890k|            continue;
 2481|       |
 2482|  54.8k|        if (f->frame_hdr->width[0] != f->frame_hdr->width[1]) {
  ------------------
  |  Branch (2482:13): [True: 15.9k, False: 38.9k]
  ------------------
 2483|  15.9k|            const int ss_hor = p && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (2483:32): [True: 10.4k, False: 5.47k]
  |  Branch (2483:37): [True: 7.30k, False: 3.14k]
  ------------------
 2484|  15.9k|            const int d = f->frame_hdr->super_res.width_scale_denominator;
 2485|  15.9k|            const int unit_size_log2 = f->frame_hdr->restoration.unit_size[!!p];
 2486|  15.9k|            const int rnd = (8 << unit_size_log2) - 1, shift = unit_size_log2 + 3;
 2487|  15.9k|            const int x = ((4 * ts->tiling.col_start * d >> ss_hor) + rnd) >> shift;
 2488|  15.9k|            const int px_x = x << (unit_size_log2 + ss_hor);
 2489|  15.9k|            const int u_idx = unit_idx + ((px_x & 64) >> 6);
 2490|  15.9k|            const int sb128x = px_x >> 7;
 2491|  15.9k|            if (sb128x >= f->sr_sb128w) continue;
  ------------------
  |  Branch (2491:17): [True: 321, False: 15.6k]
  ------------------
 2492|  15.6k|            ts->lr_ref[p] = &f->lf.lr_mask[sb_idx + sb128x].lr[p][u_idx];
 2493|  38.9k|        } else {
 2494|  38.9k|            ts->lr_ref[p] = &f->lf.lr_mask[sb_idx].lr[p][unit_idx];
 2495|  38.9k|        }
 2496|       |
 2497|  54.5k|        ts->lr_ref[p]->filter_v[0] = 3;
 2498|  54.5k|        ts->lr_ref[p]->filter_v[1] = -7;
 2499|  54.5k|        ts->lr_ref[p]->filter_v[2] = 15;
 2500|  54.5k|        ts->lr_ref[p]->filter_h[0] = 3;
 2501|  54.5k|        ts->lr_ref[p]->filter_h[1] = -7;
 2502|  54.5k|        ts->lr_ref[p]->filter_h[2] = 15;
 2503|  54.5k|        ts->lr_ref[p]->sgr_weights[0] = -32;
 2504|  54.5k|        ts->lr_ref[p]->sgr_weights[1] = 31;
 2505|  54.5k|    }
 2506|       |
 2507|   315k|    if (f->c->n_tc > 1) {
  ------------------
  |  Branch (2507:9): [True: 315k, False: 361]
  ------------------
 2508|   945k|        for (int p = 0; p < 2; p++)
  ------------------
  |  Branch (2508:25): [True: 630k, False: 315k]
  ------------------
 2509|   630k|            atomic_init(&ts->progress[p], row_sb_start);
 2510|   315k|    }
 2511|   315k|}
decode.c:get_upscale_x0:
 3324|  78.2k|static int get_upscale_x0(const int in_w, const int out_w, const int step) {
 3325|  78.2k|    const int err = out_w * step - (in_w << 14);
 3326|  78.2k|    const int x0 = (-((out_w - in_w) << 13) + (out_w >> 1)) / out_w + 128 - (err / 2);
 3327|  78.2k|    return x0 & 0x3fff;
 3328|  78.2k|}

obu.c:get_poc_diff:
  239|  1.23M|{
  240|  1.23M|    if (!order_hint_n_bits) return 0;
  ------------------
  |  Branch (240:9): [True: 0, False: 1.23M]
  ------------------
  241|  1.23M|    const int mask = 1 << (order_hint_n_bits - 1);
  242|  1.23M|    const int diff = poc0 - poc1;
  243|  1.23M|    return (diff & (mask - 1)) - (diff & mask);
  244|  1.23M|}
refmvs.c:get_gmv_2d:
  482|  4.33M|{
  483|  4.33M|    switch (gmv->type) {
  484|  1.09M|    case DAV1D_WM_TYPE_ROT_ZOOM:
  ------------------
  |  Branch (484:5): [True: 1.09M, False: 3.23M]
  ------------------
  485|  1.09M|        assert(gmv->matrix[5] ==  gmv->matrix[2]);
  ------------------
  |  Branch (485:9): [True: 1.09M, False: 18.4E]
  ------------------
  486|  1.09M|        assert(gmv->matrix[4] == -gmv->matrix[3]);
  ------------------
  |  Branch (486:9): [True: 1.09M, False: 18.4E]
  ------------------
  487|       |        // fall-through
  488|  1.09M|    default:
  ------------------
  |  Branch (488:5): [True: 0, False: 4.33M]
  ------------------
  489|  1.38M|    case DAV1D_WM_TYPE_AFFINE: {
  ------------------
  |  Branch (489:5): [True: 289k, False: 4.04M]
  ------------------
  490|  1.38M|        const int x = bx4 * 4 + bw4 * 2 - 1;
  491|  1.38M|        const int y = by4 * 4 + bh4 * 2 - 1;
  492|  1.38M|        const int xc = (gmv->matrix[2] - (1 << 16)) * x +
  493|  1.38M|                       gmv->matrix[3] * y + gmv->matrix[0];
  494|  1.38M|        const int yc = (gmv->matrix[5] - (1 << 16)) * y +
  495|  1.38M|                       gmv->matrix[4] * x + gmv->matrix[1];
  496|  1.38M|        const int shift = 16 - (3 - !hdr->hp);
  497|  1.38M|        const int round = (1 << shift) >> 1;
  498|  1.38M|        mv res = (mv) {
  499|  1.38M|            .y = apply_sign(((abs(yc) + round) >> shift) << !hdr->hp, yc),
  500|  1.38M|            .x = apply_sign(((abs(xc) + round) >> shift) << !hdr->hp, xc),
  501|  1.38M|        };
  502|  1.38M|        if (hdr->force_integer_mv)
  ------------------
  |  Branch (502:13): [True: 29.9k, False: 1.35M]
  ------------------
  503|  29.9k|            fix_int_mv_precision(&res);
  504|  1.38M|        return res;
  505|  1.09M|    }
  506|  88.6k|    case DAV1D_WM_TYPE_TRANSLATION: {
  ------------------
  |  Branch (506:5): [True: 88.6k, False: 4.24M]
  ------------------
  507|  88.6k|        mv res = (mv) {
  508|  88.6k|            .y = gmv->matrix[0] >> 13,
  509|  88.6k|            .x = gmv->matrix[1] >> 13,
  510|  88.6k|        };
  511|  88.6k|        if (hdr->force_integer_mv)
  ------------------
  |  Branch (511:13): [True: 1.29k, False: 87.3k]
  ------------------
  512|  1.29k|            fix_int_mv_precision(&res);
  513|  88.6k|        return res;
  514|  1.09M|    }
  515|  2.86M|    case DAV1D_WM_TYPE_IDENTITY:
  ------------------
  |  Branch (515:5): [True: 2.86M, False: 1.47M]
  ------------------
  516|  2.86M|        return (mv) { .x = 0, .y = 0 };
  517|  4.33M|    }
  518|  4.33M|}
refmvs.c:fix_int_mv_precision:
  462|  37.3k|static inline void fix_int_mv_precision(mv *const mv) {
  463|  37.3k|    mv->x = (mv->x - (mv->x >> 15) + 3) & ~7U;
  464|  37.3k|    mv->y = (mv->y - (mv->y >> 15) + 3) & ~7U;
  465|  37.3k|}
refmvs.c:fix_mv_precision:
  469|  2.92M|{
  470|  2.92M|    if (hdr->force_integer_mv) {
  ------------------
  |  Branch (470:9): [True: 6.10k, False: 2.92M]
  ------------------
  471|  6.10k|        fix_int_mv_precision(mv);
  472|  2.92M|    } else if (!hdr->hp) {
  ------------------
  |  Branch (472:16): [True: 301k, False: 2.62M]
  ------------------
  473|   301k|        mv->x = (mv->x - (mv->x >> 15)) & ~1U;
  474|   301k|        mv->y = (mv->y - (mv->y >> 15)) & ~1U;
  475|   301k|    }
  476|  2.92M|}
refmvs.c:get_poc_diff:
  239|  5.89M|{
  240|  5.89M|    if (!order_hint_n_bits) return 0;
  ------------------
  |  Branch (240:9): [True: 3.22M, False: 2.67M]
  ------------------
  241|  2.67M|    const int mask = 1 << (order_hint_n_bits - 1);
  242|  2.67M|    const int diff = poc0 - poc1;
  243|  2.67M|    return (diff & (mask - 1)) - (diff & mask);
  244|  5.89M|}
decode.c:get_partition_ctx:
   87|  8.89M|{
   88|  8.89M|    return ((a->partition[xb8] >> (4 - bl)) & 1) +
   89|  8.89M|          (((l->partition[yb8] >> (4 - bl)) & 1) << 1);
   90|  8.89M|}
decode.c:get_cur_frame_segid:
  445|  2.46M|{
  446|  2.46M|    cur_seg_map += bx + by * stride;
  447|  2.46M|    if (have_left && have_top) {
  ------------------
  |  Branch (447:9): [True: 1.29M, False: 1.16M]
  |  Branch (447:22): [True: 1.10M, False: 193k]
  ------------------
  448|  1.10M|        const int l = cur_seg_map[-1];
  449|  1.10M|        const int a = cur_seg_map[-stride];
  450|  1.10M|        const int al = cur_seg_map[-(stride + 1)];
  451|       |
  452|  1.10M|        if (l == a && al == l) *seg_ctx = 2;
  ------------------
  |  Branch (452:13): [True: 804k, False: 298k]
  |  Branch (452:23): [True: 786k, False: 17.9k]
  ------------------
  453|   316k|        else if (l == a || al == l || a == al) *seg_ctx = 1;
  ------------------
  |  Branch (453:18): [True: 16.9k, False: 299k]
  |  Branch (453:28): [True: 130k, False: 169k]
  |  Branch (453:39): [True: 108k, False: 61.3k]
  ------------------
  454|  60.0k|        else *seg_ctx = 0;
  455|  1.10M|        return a == al ? a : l;
  ------------------
  |  Branch (455:16): [True: 894k, False: 208k]
  ------------------
  456|  1.35M|    } else {
  457|  1.35M|        *seg_ctx = 0;
  458|  1.35M|        return have_left ? cur_seg_map[-1] : have_top ? cur_seg_map[-stride] : 0;
  ------------------
  |  Branch (458:16): [True: 195k, False: 1.16M]
  |  Branch (458:46): [True: 1.15M, False: 8.82k]
  ------------------
  459|  1.35M|    }
  460|  2.46M|}
decode.c:get_intra_ctx:
   63|  2.37M|{
   64|  2.37M|    if (have_left) {
  ------------------
  |  Branch (64:9): [True: 2.27M, False: 99.0k]
  ------------------
   65|  2.27M|        if (have_top) {
  ------------------
  |  Branch (65:13): [True: 2.02M, False: 252k]
  ------------------
   66|  2.02M|            const int ctx = l->intra[yb4] + a->intra[xb4];
   67|  2.02M|            return ctx + (ctx == 2);
   68|  2.02M|        } else
   69|   252k|            return l->intra[yb4] * 2;
   70|  2.27M|    } else {
   71|  99.0k|        return have_top ? a->intra[xb4] * 2 : 0;
  ------------------
  |  Branch (71:16): [True: 79.6k, False: 19.4k]
  ------------------
   72|  99.0k|    }
   73|  2.37M|}
decode.c:get_tx_ctx:
   79|   933k|{
   80|   933k|    return (l->tx_intra[yb4] >= max_tx->lh) + (a->tx_intra[xb4] >= max_tx->lw);
   81|   933k|}
decode.c:get_comp_ctx:
  160|   786k|{
  161|   786k|    if (have_top) {
  ------------------
  |  Branch (161:9): [True: 648k, False: 138k]
  ------------------
  162|   648k|        if (have_left) {
  ------------------
  |  Branch (162:13): [True: 613k, False: 35.2k]
  ------------------
  163|   613k|            if (a->comp_type[xb4]) {
  ------------------
  |  Branch (163:17): [True: 255k, False: 358k]
  ------------------
  164|   255k|                if (l->comp_type[yb4]) {
  ------------------
  |  Branch (164:21): [True: 164k, False: 91.0k]
  ------------------
  165|   164k|                    return 4;
  166|   164k|                } else {
  167|       |                    // 4U means intra (-1) or bwd (>= 4)
  168|  91.0k|                    return 2 + ((unsigned)l->ref[0][yb4] >= 4U);
  169|  91.0k|                }
  170|   358k|            } else if (l->comp_type[yb4]) {
  ------------------
  |  Branch (170:24): [True: 80.4k, False: 277k]
  ------------------
  171|       |                // 4U means intra (-1) or bwd (>= 4)
  172|  80.4k|                return 2 + ((unsigned)a->ref[0][xb4] >= 4U);
  173|   277k|            } else {
  174|   277k|                return (l->ref[0][yb4] >= 4) ^ (a->ref[0][xb4] >= 4);
  175|   277k|            }
  176|   613k|        } else {
  177|  35.2k|            return a->comp_type[xb4] ? 3 : a->ref[0][xb4] >= 4;
  ------------------
  |  Branch (177:20): [True: 11.8k, False: 23.3k]
  ------------------
  178|  35.2k|        }
  179|   648k|    } else if (have_left) {
  ------------------
  |  Branch (179:16): [True: 132k, False: 5.81k]
  ------------------
  180|   132k|        return l->comp_type[yb4] ? 3 : l->ref[0][yb4] >= 4;
  ------------------
  |  Branch (180:16): [True: 70.0k, False: 62.3k]
  ------------------
  181|   132k|    } else {
  182|  5.81k|        return 1;
  183|  5.81k|    }
  184|   786k|}
decode.c:fix_mv_precision:
  469|  1.45M|{
  470|  1.45M|    if (hdr->force_integer_mv) {
  ------------------
  |  Branch (470:9): [True: 16.1k, False: 1.43M]
  ------------------
  471|  16.1k|        fix_int_mv_precision(mv);
  472|  1.43M|    } else if (!hdr->hp) {
  ------------------
  |  Branch (472:16): [True: 661k, False: 772k]
  ------------------
  473|   661k|        mv->x = (mv->x - (mv->x >> 15)) & ~1U;
  474|   661k|        mv->y = (mv->y - (mv->y >> 15)) & ~1U;
  475|   661k|    }
  476|  1.45M|}
decode.c:fix_int_mv_precision:
  462|  32.8k|static inline void fix_int_mv_precision(mv *const mv) {
  463|  32.8k|    mv->x = (mv->x - (mv->x >> 15) + 3) & ~7U;
  464|  32.8k|    mv->y = (mv->y - (mv->y >> 15) + 3) & ~7U;
  465|  32.8k|}
decode.c:get_comp_dir_ctx:
  190|   372k|{
  191|   372k|#define has_uni_comp(edge, off) \
  192|   372k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  193|       |
  194|   372k|    if (have_top && have_left) {
  ------------------
  |  Branch (194:9): [True: 295k, False: 77.3k]
  |  Branch (194:21): [True: 282k, False: 12.4k]
  ------------------
  195|   282k|        const int a_intra = a->intra[xb4], l_intra = l->intra[yb4];
  196|       |
  197|   282k|        if (a_intra && l_intra) return 2;
  ------------------
  |  Branch (197:13): [True: 12.2k, False: 270k]
  |  Branch (197:24): [True: 4.77k, False: 7.43k]
  ------------------
  198|   278k|        if (a_intra || l_intra) {
  ------------------
  |  Branch (198:13): [True: 7.42k, False: 270k]
  |  Branch (198:24): [True: 9.66k, False: 261k]
  ------------------
  199|  17.1k|            const BlockContext *const edge = a_intra ? l : a;
  ------------------
  |  Branch (199:46): [True: 7.43k, False: 9.68k]
  ------------------
  200|  17.1k|            const int off = a_intra ? yb4 : xb4;
  ------------------
  |  Branch (200:29): [True: 7.43k, False: 9.68k]
  ------------------
  201|       |
  202|  17.1k|            if (edge->comp_type[off] == COMP_INTER_NONE) return 2;
  ------------------
  |  Branch (202:17): [True: 5.99k, False: 11.1k]
  ------------------
  203|  11.1k|            return 1 + 2 * has_uni_comp(edge, off);
  ------------------
  |  |  192|  11.1k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
  204|  17.1k|        }
  205|       |
  206|   260k|        const int a_comp = a->comp_type[xb4] != COMP_INTER_NONE;
  207|   260k|        const int l_comp = l->comp_type[yb4] != COMP_INTER_NONE;
  208|   260k|        const int a_ref0 = a->ref[0][xb4], l_ref0 = l->ref[0][yb4];
  209|       |
  210|   260k|        if (!a_comp && !l_comp) {
  ------------------
  |  Branch (210:13): [True: 68.8k, False: 192k]
  |  Branch (210:24): [True: 27.2k, False: 41.6k]
  ------------------
  211|  27.2k|            return 1 + 2 * ((a_ref0 >= 4) == (l_ref0 >= 4));
  212|   233k|        } else if (!a_comp || !l_comp) {
  ------------------
  |  Branch (212:20): [True: 41.5k, False: 192k]
  |  Branch (212:31): [True: 48.0k, False: 144k]
  ------------------
  213|  89.7k|            const BlockContext *const edge = a_comp ? a : l;
  ------------------
  |  Branch (213:46): [True: 48.1k, False: 41.6k]
  ------------------
  214|  89.7k|            const int off = a_comp ? xb4 : yb4;
  ------------------
  |  Branch (214:29): [True: 48.1k, False: 41.6k]
  ------------------
  215|       |
  216|  89.7k|            if (!has_uni_comp(edge, off)) return 1;
  ------------------
  |  |  192|  89.7k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
  |  Branch (216:17): [True: 74.1k, False: 15.6k]
  ------------------
  217|  15.6k|            return 3 + ((a_ref0 >= 4) == (l_ref0 >= 4));
  218|   143k|        } else {
  219|   143k|            const int a_uni = has_uni_comp(a, xb4), l_uni = has_uni_comp(l, yb4);
  ------------------
  |  |  192|   143k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
                          const int a_uni = has_uni_comp(a, xb4), l_uni = has_uni_comp(l, yb4);
  ------------------
  |  |  192|   143k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
  220|       |
  221|   143k|            if (!a_uni && !l_uni) return 0;
  ------------------
  |  Branch (221:17): [True: 115k, False: 28.3k]
  |  Branch (221:27): [True: 106k, False: 9.08k]
  ------------------
  222|  37.3k|            if (!a_uni || !l_uni) return 2;
  ------------------
  |  Branch (222:17): [True: 8.83k, False: 28.5k]
  |  Branch (222:27): [True: 15.3k, False: 13.2k]
  ------------------
  223|  12.9k|            return 3 + ((a_ref0 == 4) == (l_ref0 == 4));
  224|  37.3k|        }
  225|   260k|    } else if (have_top || have_left) {
  ------------------
  |  Branch (225:16): [True: 12.2k, False: 77.5k]
  |  Branch (225:28): [True: 75.6k, False: 1.87k]
  ------------------
  226|  88.1k|        const BlockContext *const edge = have_left ? l : a;
  ------------------
  |  Branch (226:42): [True: 75.6k, False: 12.5k]
  ------------------
  227|  88.1k|        const int off = have_left ? yb4 : xb4;
  ------------------
  |  Branch (227:25): [True: 75.6k, False: 12.5k]
  ------------------
  228|       |
  229|  88.1k|        if (edge->intra[off]) return 2;
  ------------------
  |  Branch (229:13): [True: 2.08k, False: 86.0k]
  ------------------
  230|  86.0k|        if (edge->comp_type[off] == COMP_INTER_NONE) return 2;
  ------------------
  |  Branch (230:13): [True: 17.8k, False: 68.1k]
  ------------------
  231|  68.1k|        return 4 * has_uni_comp(edge, off);
  ------------------
  |  |  192|  68.1k|    ((edge->ref[0][off] < 4) == (edge->ref[1][off] < 4))
  ------------------
  232|  86.0k|    } else {
  233|  1.63k|        return 2;
  234|  1.63k|    }
  235|   372k|}
decode.c:av1_get_fwd_ref_ctx:
  307|  1.32M|{
  308|  1.32M|    int cnt[4] = { 0 };
  309|       |
  310|  1.32M|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (310:9): [True: 1.19M, False: 129k]
  |  Branch (310:21): [True: 1.10M, False: 96.5k]
  ------------------
  311|  1.10M|        if (a->ref[0][xb4] < 4) cnt[a->ref[0][xb4]]++;
  ------------------
  |  Branch (311:13): [True: 1.02M, False: 78.1k]
  ------------------
  312|  1.10M|        if (a->comp_type[xb4] && a->ref[1][xb4] < 4) cnt[a->ref[1][xb4]]++;
  ------------------
  |  Branch (312:13): [True: 236k, False: 866k]
  |  Branch (312:34): [True: 31.9k, False: 204k]
  ------------------
  313|  1.10M|    }
  314|       |
  315|  1.32M|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (315:9): [True: 1.28M, False: 43.4k]
  |  Branch (315:22): [True: 1.18M, False: 98.3k]
  ------------------
  316|  1.18M|        if (l->ref[0][yb4] < 4) cnt[l->ref[0][yb4]]++;
  ------------------
  |  Branch (316:13): [True: 1.10M, False: 85.5k]
  ------------------
  317|  1.18M|        if (l->comp_type[yb4] && l->ref[1][yb4] < 4) cnt[l->ref[1][yb4]]++;
  ------------------
  |  Branch (317:13): [True: 282k, False: 904k]
  |  Branch (317:34): [True: 29.8k, False: 252k]
  ------------------
  318|  1.18M|    }
  319|       |
  320|  1.32M|    cnt[0] += cnt[1];
  321|  1.32M|    cnt[2] += cnt[3];
  322|       |
  323|  1.32M|    return cnt[0] == cnt[2] ? 1 : cnt[0] < cnt[2] ? 0 : 2;
  ------------------
  |  Branch (323:12): [True: 174k, False: 1.15M]
  |  Branch (323:35): [True: 183k, False: 971k]
  ------------------
  324|  1.32M|}
decode.c:av1_get_fwd_ref_2_ctx:
  350|   278k|{
  351|   278k|    int cnt[2] = { 0 };
  352|       |
  353|   278k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (353:9): [True: 221k, False: 56.9k]
  |  Branch (353:21): [True: 199k, False: 22.2k]
  ------------------
  354|   199k|        if ((a->ref[0][xb4] ^ 2U) < 2) cnt[a->ref[0][xb4] - 2]++;
  ------------------
  |  Branch (354:13): [True: 119k, False: 80.1k]
  ------------------
  355|   199k|        if (a->comp_type[xb4] && (a->ref[1][xb4] ^ 2U) < 2) cnt[a->ref[1][xb4] - 2]++;
  ------------------
  |  Branch (355:13): [True: 78.1k, False: 121k]
  |  Branch (355:34): [True: 14.0k, False: 64.0k]
  ------------------
  356|   199k|    }
  357|       |
  358|   278k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (358:9): [True: 267k, False: 11.5k]
  |  Branch (358:22): [True: 246k, False: 20.4k]
  ------------------
  359|   246k|        if ((l->ref[0][yb4] ^ 2U) < 2) cnt[l->ref[0][yb4] - 2]++;
  ------------------
  |  Branch (359:13): [True: 152k, False: 94.2k]
  ------------------
  360|   246k|        if (l->comp_type[yb4] && (l->ref[1][yb4] ^ 2U) < 2) cnt[l->ref[1][yb4] - 2]++;
  ------------------
  |  Branch (360:13): [True: 112k, False: 134k]
  |  Branch (360:34): [True: 16.9k, False: 95.3k]
  ------------------
  361|   246k|    }
  362|       |
  363|   278k|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (363:12): [True: 73.9k, False: 204k]
  |  Branch (363:35): [True: 150k, False: 54.5k]
  ------------------
  364|   278k|}
decode.c:av1_get_fwd_ref_1_ctx:
  330|  1.08M|{
  331|  1.08M|    int cnt[2] = { 0 };
  332|       |
  333|  1.08M|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (333:9): [True: 1.00M, False: 83.4k]
  |  Branch (333:21): [True: 925k, False: 75.9k]
  ------------------
  334|   925k|        if (a->ref[0][xb4] < 2) cnt[a->ref[0][xb4]]++;
  ------------------
  |  Branch (334:13): [True: 825k, False: 99.1k]
  ------------------
  335|   925k|        if (a->comp_type[xb4] && a->ref[1][xb4] < 2) cnt[a->ref[1][xb4]]++;
  ------------------
  |  Branch (335:13): [True: 173k, False: 751k]
  |  Branch (335:34): [True: 12.5k, False: 160k]
  ------------------
  336|   925k|    }
  337|       |
  338|  1.08M|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (338:9): [True: 1.05M, False: 33.2k]
  |  Branch (338:22): [True: 972k, False: 79.0k]
  ------------------
  339|   972k|        if (l->ref[0][yb4] < 2) cnt[l->ref[0][yb4]]++;
  ------------------
  |  Branch (339:13): [True: 864k, False: 107k]
  ------------------
  340|   972k|        if (l->comp_type[yb4] && l->ref[1][yb4] < 2) cnt[l->ref[1][yb4]]++;
  ------------------
  |  Branch (340:13): [True: 195k, False: 777k]
  |  Branch (340:34): [True: 11.8k, False: 183k]
  ------------------
  341|   972k|    }
  342|       |
  343|  1.08M|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (343:12): [True: 132k, False: 952k]
  |  Branch (343:35): [True: 56.0k, False: 896k]
  ------------------
  344|  1.08M|}
decode.c:av1_get_bwd_ref_ctx:
  370|   826k|{
  371|   826k|    int cnt[3] = { 0 };
  372|       |
  373|   826k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (373:9): [True: 667k, False: 159k]
  |  Branch (373:21): [True: 624k, False: 42.9k]
  ------------------
  374|   624k|        if (a->ref[0][xb4] >= 4) cnt[a->ref[0][xb4] - 4]++;
  ------------------
  |  Branch (374:13): [True: 342k, False: 282k]
  ------------------
  375|   624k|        if (a->comp_type[xb4] && a->ref[1][xb4] >= 4) cnt[a->ref[1][xb4] - 4]++;
  ------------------
  |  Branch (375:13): [True: 210k, False: 413k]
  |  Branch (375:34): [True: 195k, False: 15.7k]
  ------------------
  376|   624k|    }
  377|       |
  378|   826k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (378:9): [True: 787k, False: 39.7k]
  |  Branch (378:22): [True: 733k, False: 53.4k]
  ------------------
  379|   733k|        if (l->ref[0][yb4] >= 4) cnt[l->ref[0][yb4] - 4]++;
  ------------------
  |  Branch (379:13): [True: 397k, False: 336k]
  ------------------
  380|   733k|        if (l->comp_type[yb4] && l->ref[1][yb4] >= 4) cnt[l->ref[1][yb4] - 4]++;
  ------------------
  |  Branch (380:13): [True: 253k, False: 479k]
  |  Branch (380:34): [True: 238k, False: 14.9k]
  ------------------
  381|   733k|    }
  382|       |
  383|   826k|    cnt[1] += cnt[0];
  384|       |
  385|   826k|    return cnt[2] == cnt[1] ? 1 : cnt[1] < cnt[2] ? 0 : 2;
  ------------------
  |  Branch (385:12): [True: 145k, False: 681k]
  |  Branch (385:35): [True: 462k, False: 219k]
  ------------------
  386|   826k|}
decode.c:av1_get_bwd_ref_1_ctx:
  392|   285k|{
  393|   285k|    int cnt[3] = { 0 };
  394|       |
  395|   285k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (395:9): [True: 252k, False: 32.7k]
  |  Branch (395:21): [True: 233k, False: 19.1k]
  ------------------
  396|   233k|        if (a->ref[0][xb4] >= 4) cnt[a->ref[0][xb4] - 4]++;
  ------------------
  |  Branch (396:13): [True: 97.9k, False: 135k]
  ------------------
  397|   233k|        if (a->comp_type[xb4] && a->ref[1][xb4] >= 4) cnt[a->ref[1][xb4] - 4]++;
  ------------------
  |  Branch (397:13): [True: 102k, False: 131k]
  |  Branch (397:34): [True: 92.0k, False: 10.2k]
  ------------------
  398|   233k|    }
  399|       |
  400|   285k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (400:9): [True: 275k, False: 9.79k]
  |  Branch (400:22): [True: 253k, False: 22.3k]
  ------------------
  401|   253k|        if (l->ref[0][yb4] >= 4) cnt[l->ref[0][yb4] - 4]++;
  ------------------
  |  Branch (401:13): [True: 105k, False: 148k]
  ------------------
  402|   253k|        if (l->comp_type[yb4] && l->ref[1][yb4] >= 4) cnt[l->ref[1][yb4] - 4]++;
  ------------------
  |  Branch (402:13): [True: 112k, False: 141k]
  |  Branch (402:34): [True: 102k, False: 9.80k]
  ------------------
  403|   253k|    }
  404|       |
  405|   285k|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (405:12): [True: 64.6k, False: 220k]
  |  Branch (405:35): [True: 76.0k, False: 144k]
  ------------------
  406|   285k|}
decode.c:av1_get_ref_ctx:
  287|  1.62M|{
  288|  1.62M|    int cnt[2] = { 0 };
  289|       |
  290|  1.62M|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (290:9): [True: 1.43M, False: 183k]
  |  Branch (290:21): [True: 1.31M, False: 124k]
  ------------------
  291|  1.31M|        cnt[a->ref[0][xb4] >= 4]++;
  292|  1.31M|        if (a->comp_type[xb4]) cnt[a->ref[1][xb4] >= 4]++;
  ------------------
  |  Branch (292:13): [True: 137k, False: 1.17M]
  ------------------
  293|  1.31M|    }
  294|       |
  295|  1.62M|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (295:9): [True: 1.55M, False: 62.9k]
  |  Branch (295:22): [True: 1.42M, False: 131k]
  ------------------
  296|  1.42M|        cnt[l->ref[0][yb4] >= 4]++;
  297|  1.42M|        if (l->comp_type[yb4]) cnt[l->ref[1][yb4] >= 4]++;
  ------------------
  |  Branch (297:13): [True: 177k, False: 1.24M]
  ------------------
  298|  1.42M|    }
  299|       |
  300|  1.62M|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (300:12): [True: 219k, False: 1.40M]
  |  Branch (300:35): [True: 440k, False: 960k]
  ------------------
  301|  1.62M|}
decode.c:av1_get_uni_p1_ctx:
  412|  52.3k|{
  413|  52.3k|    int cnt[3] = { 0 };
  414|       |
  415|  52.3k|    if (have_top && !a->intra[xb4]) {
  ------------------
  |  Branch (415:9): [True: 40.7k, False: 11.5k]
  |  Branch (415:21): [True: 37.7k, False: 3.05k]
  ------------------
  416|  37.7k|        if (a->ref[0][xb4] - 1U < 3) cnt[a->ref[0][xb4] - 1]++;
  ------------------
  |  Branch (416:13): [True: 5.58k, False: 32.1k]
  ------------------
  417|  37.7k|        if (a->comp_type[xb4] && a->ref[1][xb4] - 1U < 3) cnt[a->ref[1][xb4] - 1]++;
  ------------------
  |  Branch (417:13): [True: 27.2k, False: 10.4k]
  |  Branch (417:34): [True: 16.0k, False: 11.1k]
  ------------------
  418|  37.7k|    }
  419|       |
  420|  52.3k|    if (have_left && !l->intra[yb4]) {
  ------------------
  |  Branch (420:9): [True: 50.4k, False: 1.89k]
  |  Branch (420:22): [True: 48.5k, False: 1.93k]
  ------------------
  421|  48.5k|        if (l->ref[0][yb4] - 1U < 3) cnt[l->ref[0][yb4] - 1]++;
  ------------------
  |  Branch (421:13): [True: 9.35k, False: 39.1k]
  ------------------
  422|  48.5k|        if (l->comp_type[yb4] && l->ref[1][yb4] - 1U < 3) cnt[l->ref[1][yb4] - 1]++;
  ------------------
  |  Branch (422:13): [True: 37.2k, False: 11.2k]
  |  Branch (422:34): [True: 20.3k, False: 16.9k]
  ------------------
  423|  48.5k|    }
  424|       |
  425|  52.3k|    cnt[1] += cnt[2];
  426|       |
  427|  52.3k|    return cnt[0] == cnt[1] ? 1 : cnt[0] < cnt[1] ? 0 : 2;
  ------------------
  |  Branch (427:12): [True: 16.7k, False: 35.5k]
  |  Branch (427:35): [True: 23.4k, False: 12.1k]
  ------------------
  428|  52.3k|}
decode.c:get_drl_context:
  432|   852k|{
  433|   852k|    if (ref_mv_stack[ref_idx].weight >= 640)
  ------------------
  |  Branch (433:9): [True: 702k, False: 149k]
  ------------------
  434|   702k|        return ref_mv_stack[ref_idx + 1].weight < 640;
  435|       |
  436|  18.4E|    return ref_mv_stack[ref_idx + 1].weight < 640 ? 2 : 0;
  ------------------
  |  Branch (436:12): [True: 149k, False: 18.4E]
  ------------------
  437|   852k|}
decode.c:get_gmv_2d:
  482|  2.14M|{
  483|  2.14M|    switch (gmv->type) {
  484|   783k|    case DAV1D_WM_TYPE_ROT_ZOOM:
  ------------------
  |  Branch (484:5): [True: 783k, False: 1.35M]
  ------------------
  485|   783k|        assert(gmv->matrix[5] ==  gmv->matrix[2]);
  ------------------
  |  Branch (485:9): [True: 783k, False: 18.4E]
  ------------------
  486|   783k|        assert(gmv->matrix[4] == -gmv->matrix[3]);
  ------------------
  |  Branch (486:9): [True: 783k, False: 6]
  ------------------
  487|       |        // fall-through
  488|   783k|    default:
  ------------------
  |  Branch (488:5): [True: 0, False: 2.14M]
  ------------------
  489|   918k|    case DAV1D_WM_TYPE_AFFINE: {
  ------------------
  |  Branch (489:5): [True: 135k, False: 2.00M]
  ------------------
  490|   918k|        const int x = bx4 * 4 + bw4 * 2 - 1;
  491|   918k|        const int y = by4 * 4 + bh4 * 2 - 1;
  492|   918k|        const int xc = (gmv->matrix[2] - (1 << 16)) * x +
  493|   918k|                       gmv->matrix[3] * y + gmv->matrix[0];
  494|   918k|        const int yc = (gmv->matrix[5] - (1 << 16)) * y +
  495|   918k|                       gmv->matrix[4] * x + gmv->matrix[1];
  496|   918k|        const int shift = 16 - (3 - !hdr->hp);
  497|   918k|        const int round = (1 << shift) >> 1;
  498|   918k|        mv res = (mv) {
  499|   918k|            .y = apply_sign(((abs(yc) + round) >> shift) << !hdr->hp, yc),
  500|   918k|            .x = apply_sign(((abs(xc) + round) >> shift) << !hdr->hp, xc),
  501|   918k|        };
  502|   918k|        if (hdr->force_integer_mv)
  ------------------
  |  Branch (502:13): [True: 16.2k, False: 902k]
  ------------------
  503|  16.2k|            fix_int_mv_precision(&res);
  504|   918k|        return res;
  505|   783k|    }
  506|  65.1k|    case DAV1D_WM_TYPE_TRANSLATION: {
  ------------------
  |  Branch (506:5): [True: 65.1k, False: 2.07M]
  ------------------
  507|  65.1k|        mv res = (mv) {
  508|  65.1k|            .y = gmv->matrix[0] >> 13,
  509|  65.1k|            .x = gmv->matrix[1] >> 13,
  510|  65.1k|        };
  511|  65.1k|        if (hdr->force_integer_mv)
  ------------------
  |  Branch (511:13): [True: 441, False: 64.7k]
  ------------------
  512|    441|            fix_int_mv_precision(&res);
  513|  65.1k|        return res;
  514|   783k|    }
  515|  1.15M|    case DAV1D_WM_TYPE_IDENTITY:
  ------------------
  |  Branch (515:5): [True: 1.15M, False: 983k]
  ------------------
  516|  1.15M|        return (mv) { .x = 0, .y = 0 };
  517|  2.14M|    }
  518|  2.14M|}
decode.c:get_mask_comp_ctx:
  266|   295k|{
  267|   295k|    const int a_ctx = a->comp_type[xb4] >= COMP_INTER_SEG ? 1 :
  ------------------
  |  Branch (267:23): [True: 38.9k, False: 256k]
  ------------------
  268|   295k|                      a->ref[0][xb4] == 6 ? 3 : 0;
  ------------------
  |  Branch (268:23): [True: 14.8k, False: 241k]
  ------------------
  269|   295k|    const int l_ctx = l->comp_type[yb4] >= COMP_INTER_SEG ? 1 :
  ------------------
  |  Branch (269:23): [True: 43.6k, False: 251k]
  ------------------
  270|   295k|                      l->ref[0][yb4] == 6 ? 3 : 0;
  ------------------
  |  Branch (270:23): [True: 18.0k, False: 233k]
  ------------------
  271|       |
  272|   295k|    return imin(a_ctx + l_ctx, 5);
  273|   295k|}
decode.c:get_jnt_comp_ctx:
  251|   228k|{
  252|   228k|    const int d0 = abs(get_poc_diff(order_hint_n_bits, ref0poc, poc));
  253|   228k|    const int d1 = abs(get_poc_diff(order_hint_n_bits, poc, ref1poc));
  254|   228k|    const int offset = d0 == d1;
  255|   228k|    const int a_ctx = a->comp_type[xb4] >= COMP_INTER_AVG ||
  ------------------
  |  Branch (255:23): [True: 100k, False: 128k]
  ------------------
  256|   128k|                      a->ref[0][xb4] == 6;
  ------------------
  |  Branch (256:23): [True: 10.3k, False: 117k]
  ------------------
  257|   228k|    const int l_ctx = l->comp_type[yb4] >= COMP_INTER_AVG ||
  ------------------
  |  Branch (257:23): [True: 115k, False: 113k]
  ------------------
  258|   113k|                      l->ref[0][yb4] == 6;
  ------------------
  |  Branch (258:23): [True: 12.1k, False: 101k]
  ------------------
  259|       |
  260|   228k|    return 3 * offset + a_ctx + l_ctx;
  261|   228k|}
decode.c:get_filter_ctx:
  139|  1.36M|{
  140|  1.36M|    const int a_filter = (a->ref[0][xb4] == ref || a->ref[1][xb4] == ref) ?
  ------------------
  |  Branch (140:27): [True: 836k, False: 526k]
  |  Branch (140:52): [True: 42.5k, False: 484k]
  ------------------
  141|   879k|                         a->filter[dir][xb4] : DAV1D_N_SWITCHABLE_FILTERS;
  142|  1.36M|    const int l_filter = (l->ref[0][yb4] == ref || l->ref[1][yb4] == ref) ?
  ------------------
  |  Branch (142:27): [True: 930k, False: 432k]
  |  Branch (142:52): [True: 41.4k, False: 390k]
  ------------------
  143|   973k|                         l->filter[dir][yb4] : DAV1D_N_SWITCHABLE_FILTERS;
  144|       |
  145|  1.36M|    if (a_filter == l_filter) {
  ------------------
  |  Branch (145:9): [True: 812k, False: 550k]
  ------------------
  146|   812k|        return comp * 4 + a_filter;
  147|   812k|    } else if (a_filter == DAV1D_N_SWITCHABLE_FILTERS) {
  ------------------
  |  Branch (147:16): [True: 297k, False: 252k]
  ------------------
  148|   297k|        return comp * 4 + l_filter;
  149|   297k|    } else if (l_filter == DAV1D_N_SWITCHABLE_FILTERS) {
  ------------------
  |  Branch (149:16): [True: 203k, False: 48.4k]
  ------------------
  150|   203k|        return comp * 4 + a_filter;
  151|   203k|    } else {
  152|  48.4k|        return comp * 4 + DAV1D_N_SWITCHABLE_FILTERS;
  153|  48.4k|    }
  154|  1.36M|}
decode.c:gather_top_partition_prob:
  106|   362k|{
  107|       |    // Exploit the fact that cdfs for PARTITION_V, PARTITION_SPLIT and
  108|       |    // PARTITION_T_TOP_SPLIT are neighbors.
  109|   362k|    unsigned out = in[PARTITION_V - 1] - in[PARTITION_T_TOP_SPLIT];
  110|       |    // Exploit the facts that cdfs for PARTITION_T_LEFT_SPLIT and
  111|       |    // PARTITION_T_RIGHT_SPLIT are neighbors, the probability for
  112|       |    // PARTITION_V4 is always zero, and the probability for
  113|       |    // PARTITION_T_RIGHT_SPLIT is zero in 128x128 blocks.
  114|   362k|    out += in[PARTITION_T_LEFT_SPLIT - 1];
  115|   362k|    if (bl != BL_128X128)
  ------------------
  |  Branch (115:9): [True: 316k, False: 45.7k]
  ------------------
  116|   316k|        out += in[PARTITION_V4 - 1] - in[PARTITION_T_RIGHT_SPLIT];
  117|   362k|    return out;
  118|   362k|}
decode.c:gather_left_partition_prob:
   94|   371k|{
   95|   371k|    unsigned out = in[PARTITION_H - 1] - in[PARTITION_H];
   96|       |    // Exploit the fact that cdfs for PARTITION_SPLIT, PARTITION_T_TOP_SPLIT,
   97|       |    // PARTITION_T_BOTTOM_SPLIT and PARTITION_T_LEFT_SPLIT are neighbors.
   98|   371k|    out += in[PARTITION_SPLIT - 1] - in[PARTITION_T_LEFT_SPLIT];
   99|   371k|    if (bl != BL_128X128)
  ------------------
  |  Branch (99:9): [True: 335k, False: 36.2k]
  ------------------
  100|   335k|        out += in[PARTITION_H4 - 1] - in[PARTITION_H4];
  101|   371k|    return out;
  102|   371k|}
decode.c:get_poc_diff:
  239|  2.64M|{
  240|  2.64M|    if (!order_hint_n_bits) return 0;
  ------------------
  |  Branch (240:9): [True: 60.1k, False: 2.58M]
  ------------------
  241|  2.58M|    const int mask = 1 << (order_hint_n_bits - 1);
  242|  2.58M|    const int diff = poc0 - poc1;
  243|  2.58M|    return (diff & (mask - 1)) - (diff & mask);
  244|  2.64M|}
recon_tmpl.c:get_uv_inter_txtp:
  122|   852k|{
  123|   852k|    if (uvt_dim->max == TX_32X32)
  ------------------
  |  Branch (123:9): [True: 166k, False: 685k]
  ------------------
  124|   166k|        return ytxtp == IDTX ? IDTX : DCT_DCT;
  ------------------
  |  Branch (124:16): [True: 4.45k, False: 162k]
  ------------------
  125|   685k|    if (uvt_dim->min == TX_16X16 &&
  ------------------
  |  Branch (125:9): [True: 70.5k, False: 614k]
  ------------------
  126|  70.5k|        ((1 << ytxtp) & ((1 << H_FLIPADST) | (1 << V_FLIPADST) |
  ------------------
  |  Branch (126:9): [True: 582, False: 69.9k]
  ------------------
  127|  70.5k|                         (1 << H_ADST) | (1 << V_ADST))))
  128|    582|    {
  129|    582|        return DCT_DCT;
  130|    582|    }
  131|       |
  132|   684k|    return ytxtp;
  133|   685k|}

dav1d_prep_grain_8bpc:
  105|  7.70k|{
  106|  7.70k|    const Dav1dFilmGrainData *const data = &out->frame_hdr->film_grain.data;
  107|       |#if BITDEPTH != 8
  108|       |    const int bitdepth_max = (1 << out->p.bpc) - 1;
  109|       |#endif
  110|       |
  111|       |    // Generate grain LUTs as needed
  112|  7.70k|    dsp->generate_grain_y(grain_lut[0], data HIGHBD_TAIL_SUFFIX); // always needed
  113|  7.70k|    if (data->num_uv_points[0] || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (113:9): [True: 720, False: 6.98k]
  |  Branch (113:35): [True: 712, False: 6.27k]
  ------------------
  114|  1.43k|        dsp->generate_grain_uv[in->p.layout - 1](grain_lut[1], grain_lut[0],
  115|  1.43k|                                                 data, 0 HIGHBD_TAIL_SUFFIX);
  116|  7.70k|    if (data->num_uv_points[1] || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (116:9): [True: 3.21k, False: 4.49k]
  |  Branch (116:35): [True: 712, False: 3.78k]
  ------------------
  117|  3.92k|        dsp->generate_grain_uv[in->p.layout - 1](grain_lut[2], grain_lut[0],
  118|  3.92k|                                                 data, 1 HIGHBD_TAIL_SUFFIX);
  119|       |
  120|       |    // Generate scaling LUTs as needed
  121|  7.70k|    if (data->num_y_points || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (121:9): [True: 3.78k, False: 3.92k]
  |  Branch (121:31): [True: 700, False: 3.22k]
  ------------------
  122|  4.48k|        generate_scaling(in->p.bpc, data->y_points, data->num_y_points, scaling[0]);
  123|  7.70k|    if (data->num_uv_points[0])
  ------------------
  |  Branch (123:9): [True: 720, False: 6.98k]
  ------------------
  124|    720|        generate_scaling(in->p.bpc, data->uv_points[0], data->num_uv_points[0], scaling[1]);
  125|  7.70k|    if (data->num_uv_points[1])
  ------------------
  |  Branch (125:9): [True: 3.21k, False: 4.49k]
  ------------------
  126|  3.21k|        generate_scaling(in->p.bpc, data->uv_points[1], data->num_uv_points[1], scaling[2]);
  127|       |
  128|       |    // Copy over the non-modified planes
  129|  7.70k|    assert(out->stride[0] == in->stride[0]);
  ------------------
  |  Branch (129:5): [True: 7.70k, False: 0]
  ------------------
  130|  7.70k|    if (!data->num_y_points) {
  ------------------
  |  Branch (130:9): [True: 3.92k, False: 3.78k]
  ------------------
  131|  3.92k|        const ptrdiff_t stride = out->stride[0];
  132|  3.92k|        const ptrdiff_t sz = out->p.h * stride;
  133|  3.92k|        if (sz < 0)
  ------------------
  |  Branch (133:13): [True: 0, False: 3.92k]
  ------------------
  134|      0|            memcpy((uint8_t*) out->data[0] + sz - stride,
  135|      0|                   (uint8_t*) in->data[0] + sz - stride, -sz);
  136|  3.92k|        else
  137|  3.92k|            memcpy(out->data[0], in->data[0], sz);
  138|  3.92k|    }
  139|       |
  140|  7.70k|    if (in->p.layout != DAV1D_PIXEL_LAYOUT_I400 && !data->chroma_scaling_from_luma) {
  ------------------
  |  Branch (140:9): [True: 7.20k, False: 501]
  |  Branch (140:52): [True: 6.49k, False: 712]
  ------------------
  141|  6.49k|        assert(out->stride[1] == in->stride[1]);
  ------------------
  |  Branch (141:9): [True: 6.49k, False: 0]
  ------------------
  142|  6.49k|        const int ss_ver = in->p.layout == DAV1D_PIXEL_LAYOUT_I420;
  143|  6.49k|        const ptrdiff_t stride = out->stride[1];
  144|  6.49k|        const ptrdiff_t sz = ((out->p.h + ss_ver) >> ss_ver) * stride;
  145|  6.49k|        if (sz < 0) {
  ------------------
  |  Branch (145:13): [True: 0, False: 6.49k]
  ------------------
  146|      0|            if (!data->num_uv_points[0])
  ------------------
  |  Branch (146:17): [True: 0, False: 0]
  ------------------
  147|      0|                memcpy((uint8_t*) out->data[1] + sz - stride,
  148|      0|                       (uint8_t*) in->data[1] + sz - stride, -sz);
  149|      0|            if (!data->num_uv_points[1])
  ------------------
  |  Branch (149:17): [True: 0, False: 0]
  ------------------
  150|      0|                memcpy((uint8_t*) out->data[2] + sz - stride,
  151|      0|                       (uint8_t*) in->data[2] + sz - stride, -sz);
  152|  6.49k|        } else {
  153|  6.49k|            if (!data->num_uv_points[0])
  ------------------
  |  Branch (153:17): [True: 5.77k, False: 720]
  ------------------
  154|  5.77k|                memcpy(out->data[1], in->data[1], sz);
  155|  6.49k|            if (!data->num_uv_points[1])
  ------------------
  |  Branch (155:17): [True: 3.28k, False: 3.21k]
  ------------------
  156|  3.28k|                memcpy(out->data[2], in->data[2], sz);
  157|  6.49k|        }
  158|  6.49k|    }
  159|  7.70k|}
dav1d_apply_grain_row_8bpc:
  167|  29.8k|{
  168|       |    // Synthesize grain for the affected planes
  169|  29.8k|    const Dav1dFilmGrainData *const data = &out->frame_hdr->film_grain.data;
  170|  29.8k|    const int ss_y = in->p.layout == DAV1D_PIXEL_LAYOUT_I420;
  171|  29.8k|    const int ss_x = in->p.layout != DAV1D_PIXEL_LAYOUT_I444;
  172|  29.8k|    const int cpw = (out->p.w + ss_x) >> ss_x;
  173|  29.8k|    const int is_id = out->seq_hdr->mtrx == DAV1D_MC_IDENTITY;
  174|  29.8k|    pixel *const luma_src =
  175|  29.8k|        ((pixel *) in->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(in->stride[0]);
  ------------------
  |  |   37|  29.8k|#define FG_BLOCK_SIZE 32
  ------------------
                      ((pixel *) in->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(in->stride[0]);
  ------------------
  |  |   53|  29.8k|#define PXSTRIDE(x) (x)
  ------------------
  176|       |#if BITDEPTH != 8
  177|       |    const int bitdepth_max = (1 << out->p.bpc) - 1;
  178|       |#endif
  179|       |
  180|  29.8k|    if (data->num_y_points) {
  ------------------
  |  Branch (180:9): [True: 14.7k, False: 15.0k]
  ------------------
  181|  14.7k|        const int bh = imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE);
  ------------------
  |  |   37|  14.7k|#define FG_BLOCK_SIZE 32
  ------------------
                      const int bh = imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE);
  ------------------
  |  |   37|  14.7k|#define FG_BLOCK_SIZE 32
  ------------------
  182|  14.7k|        dsp->fgy_32x32xn(((pixel *) out->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[0]),
  ------------------
  |  |   37|  14.7k|#define FG_BLOCK_SIZE 32
  ------------------
                      dsp->fgy_32x32xn(((pixel *) out->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[0]),
  ------------------
  |  |   53|  14.7k|#define PXSTRIDE(x) (x)
  ------------------
  183|  14.7k|                         luma_src, out->stride[0], data,
  184|  14.7k|                         out->p.w, scaling[0], grain_lut[0], bh, row HIGHBD_TAIL_SUFFIX);
  185|  14.7k|    }
  186|       |
  187|  29.8k|    if (!data->num_uv_points[0] && !data->num_uv_points[1] &&
  ------------------
  |  Branch (187:9): [True: 25.6k, False: 4.22k]
  |  Branch (187:36): [True: 13.8k, False: 11.8k]
  ------------------
  188|  13.8k|        !data->chroma_scaling_from_luma)
  ------------------
  |  Branch (188:9): [True: 10.3k, False: 3.42k]
  ------------------
  189|  10.3k|    {
  190|  10.3k|        return;
  191|  10.3k|    }
  192|       |
  193|  19.4k|    const int bh = (imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE) + ss_y) >> ss_y;
  ------------------
  |  |   37|  19.4k|#define FG_BLOCK_SIZE 32
  ------------------
                  const int bh = (imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE) + ss_y) >> ss_y;
  ------------------
  |  |   37|  19.4k|#define FG_BLOCK_SIZE 32
  ------------------
  194|       |
  195|       |    // extend padding pixels
  196|  19.4k|    if (out->p.w & ss_x) {
  ------------------
  |  Branch (196:9): [True: 1.95k, False: 17.4k]
  ------------------
  197|  1.95k|        pixel *ptr = luma_src;
  198|  41.9k|        for (int y = 0; y < bh; y++) {
  ------------------
  |  Branch (198:25): [True: 39.9k, False: 1.95k]
  ------------------
  199|  39.9k|            ptr[out->p.w] = ptr[out->p.w - 1];
  200|  39.9k|            ptr += PXSTRIDE(in->stride[0]) << ss_y;
  ------------------
  |  |   53|  39.9k|#define PXSTRIDE(x) (x)
  ------------------
  201|  39.9k|        }
  202|  1.95k|    }
  203|       |
  204|  19.4k|    const ptrdiff_t uv_off = row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[1]) >> ss_y;
  ------------------
  |  |   37|  19.4k|#define FG_BLOCK_SIZE 32
  ------------------
                  const ptrdiff_t uv_off = row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[1]) >> ss_y;
  ------------------
  |  |   53|  19.4k|#define PXSTRIDE(x) (x)
  ------------------
  205|  19.4k|    if (data->chroma_scaling_from_luma) {
  ------------------
  |  Branch (205:9): [True: 3.35k, False: 16.0k]
  ------------------
  206|  10.0k|        for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (206:26): [True: 6.70k, False: 3.35k]
  ------------------
  207|  6.70k|            dsp->fguv_32x32xn[in->p.layout - 1](((pixel *) out->data[1 + pl]) + uv_off,
  208|  6.70k|                                                ((const pixel *) in->data[1 + pl]) + uv_off,
  209|  6.70k|                                                in->stride[1], data, cpw,
  210|  6.70k|                                                scaling[0], grain_lut[1 + pl],
  211|  6.70k|                                                bh, row, luma_src, in->stride[0],
  212|  6.70k|                                                pl, is_id HIGHBD_TAIL_SUFFIX);
  213|  16.0k|    } else {
  214|  48.1k|        for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (214:26): [True: 32.0k, False: 16.0k]
  ------------------
  215|  32.0k|            if (data->num_uv_points[pl])
  ------------------
  |  Branch (215:17): [True: 18.2k, False: 13.7k]
  ------------------
  216|  18.2k|                dsp->fguv_32x32xn[in->p.layout - 1](((pixel *) out->data[1 + pl]) + uv_off,
  217|  18.2k|                                                    ((const pixel *) in->data[1 + pl]) + uv_off,
  218|  18.2k|                                                    in->stride[1], data, cpw,
  219|  18.2k|                                                    scaling[1 + pl], grain_lut[1 + pl],
  220|  18.2k|                                                    bh, row, luma_src, in->stride[0],
  221|  18.2k|                                                    pl, is_id HIGHBD_TAIL_SUFFIX);
  222|  16.0k|    }
  223|  19.4k|}
fg_apply_tmpl.c:generate_scaling:
   44|  8.41k|{
   45|  8.41k|#if BITDEPTH == 8
   46|  8.41k|    const int shift_x = 0;
   47|  8.41k|    const int scaling_size = SCALING_SIZE;
  ------------------
  |  |   39|  8.41k|#define SCALING_SIZE 256
  ------------------
   48|       |#else
   49|       |    assert(bitdepth > 8);
   50|       |    const int shift_x = bitdepth - 8;
   51|       |    const int scaling_size = 1 << bitdepth;
   52|       |#endif
   53|       |
   54|  8.41k|    if (num == 0) {
  ------------------
  |  Branch (54:9): [True: 700, False: 7.71k]
  ------------------
   55|    700|        memset(scaling, 0, scaling_size);
   56|    700|        return;
   57|    700|    }
   58|       |
   59|       |    // Fill up the preceding entries with the initial value
   60|  7.71k|    memset(scaling, points[0][1], points[0][0] << shift_x);
   61|       |
   62|       |    // Linearly interpolate the values in the middle
   63|  19.3k|    for (int i = 0; i < num - 1; i++) {
  ------------------
  |  Branch (63:21): [True: 11.6k, False: 7.71k]
  ------------------
   64|  11.6k|        const int bx = points[i][0];
   65|  11.6k|        const int by = points[i][1];
   66|  11.6k|        const int ex = points[i+1][0];
   67|  11.6k|        const int ey = points[i+1][1];
   68|  11.6k|        const int dx = ex - bx;
   69|  11.6k|        const int dy = ey - by;
   70|  11.6k|        assert(dx > 0);
  ------------------
  |  Branch (70:9): [True: 11.6k, False: 0]
  ------------------
   71|  11.6k|        const int delta = dy * ((0x10000 + (dx >> 1)) / dx);
   72|   575k|        for (int x = 0, d = 0x8000; x < dx; x++) {
  ------------------
  |  Branch (72:37): [True: 563k, False: 11.6k]
  ------------------
   73|   563k|            scaling[(bx + x) << shift_x] = by + (d >> 16);
   74|   563k|            d += delta;
   75|   563k|        }
   76|  11.6k|    }
   77|       |
   78|       |    // Fill up the remaining entries with the final value
   79|  7.71k|    const int n = points[num - 1][0] << shift_x;
   80|  7.71k|    memset(&scaling[n], points[num - 1][1], scaling_size - n);
   81|       |
   82|       |#if BITDEPTH != 8
   83|       |    const int pad = 1 << shift_x, rnd = pad >> 1;
   84|       |    for (int i = 0; i < num - 1; i++) {
   85|       |        const int bx = points[i][0] << shift_x;
   86|       |        const int ex = points[i+1][0] << shift_x;
   87|       |        const int dx = ex - bx;
   88|       |        for (int x = 0; x < dx; x += pad) {
   89|       |            const int range = scaling[bx + x + pad] - scaling[bx + x];
   90|       |            for (int n = 1, r = rnd; n < pad; n++) {
   91|       |                r += range;
   92|       |                scaling[bx + x + n] = scaling[bx + x] + (r >> shift_x);
   93|       |            }
   94|       |        }
   95|       |    }
   96|       |#endif
   97|  7.71k|}
dav1d_prep_grain_16bpc:
  105|  2.40k|{
  106|  2.40k|    const Dav1dFilmGrainData *const data = &out->frame_hdr->film_grain.data;
  107|  2.40k|#if BITDEPTH != 8
  108|  2.40k|    const int bitdepth_max = (1 << out->p.bpc) - 1;
  109|  2.40k|#endif
  110|       |
  111|       |    // Generate grain LUTs as needed
  112|  2.40k|    dsp->generate_grain_y(grain_lut[0], data HIGHBD_TAIL_SUFFIX); // always needed
  ------------------
  |  |   74|  2.40k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  113|  2.40k|    if (data->num_uv_points[0] || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (113:9): [True: 426, False: 1.98k]
  |  Branch (113:35): [True: 1.44k, False: 533]
  ------------------
  114|  1.87k|        dsp->generate_grain_uv[in->p.layout - 1](grain_lut[1], grain_lut[0],
  115|  1.87k|                                                 data, 0 HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  1.87k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  116|  2.40k|    if (data->num_uv_points[1] || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (116:9): [True: 263, False: 2.14k]
  |  Branch (116:35): [True: 1.44k, False: 696]
  ------------------
  117|  1.71k|        dsp->generate_grain_uv[in->p.layout - 1](grain_lut[2], grain_lut[0],
  118|  1.71k|                                                 data, 1 HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  1.71k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  119|       |
  120|       |    // Generate scaling LUTs as needed
  121|  2.40k|    if (data->num_y_points || data->chroma_scaling_from_luma)
  ------------------
  |  Branch (121:9): [True: 784, False: 1.62k]
  |  Branch (121:31): [True: 1.02k, False: 595]
  ------------------
  122|  1.81k|        generate_scaling(in->p.bpc, data->y_points, data->num_y_points, scaling[0]);
  123|  2.40k|    if (data->num_uv_points[0])
  ------------------
  |  Branch (123:9): [True: 426, False: 1.98k]
  ------------------
  124|    426|        generate_scaling(in->p.bpc, data->uv_points[0], data->num_uv_points[0], scaling[1]);
  125|  2.40k|    if (data->num_uv_points[1])
  ------------------
  |  Branch (125:9): [True: 263, False: 2.14k]
  ------------------
  126|    263|        generate_scaling(in->p.bpc, data->uv_points[1], data->num_uv_points[1], scaling[2]);
  127|       |
  128|       |    // Copy over the non-modified planes
  129|  2.40k|    assert(out->stride[0] == in->stride[0]);
  ------------------
  |  Branch (129:5): [True: 2.40k, False: 0]
  ------------------
  130|  2.40k|    if (!data->num_y_points) {
  ------------------
  |  Branch (130:9): [True: 1.62k, False: 784]
  ------------------
  131|  1.62k|        const ptrdiff_t stride = out->stride[0];
  132|  1.62k|        const ptrdiff_t sz = out->p.h * stride;
  133|  1.62k|        if (sz < 0)
  ------------------
  |  Branch (133:13): [True: 0, False: 1.62k]
  ------------------
  134|      0|            memcpy((uint8_t*) out->data[0] + sz - stride,
  135|      0|                   (uint8_t*) in->data[0] + sz - stride, -sz);
  136|  1.62k|        else
  137|  1.62k|            memcpy(out->data[0], in->data[0], sz);
  138|  1.62k|    }
  139|       |
  140|  2.40k|    if (in->p.layout != DAV1D_PIXEL_LAYOUT_I400 && !data->chroma_scaling_from_luma) {
  ------------------
  |  Branch (140:9): [True: 2.22k, False: 184]
  |  Branch (140:52): [True: 775, False: 1.44k]
  ------------------
  141|    775|        assert(out->stride[1] == in->stride[1]);
  ------------------
  |  Branch (141:9): [True: 775, False: 0]
  ------------------
  142|    775|        const int ss_ver = in->p.layout == DAV1D_PIXEL_LAYOUT_I420;
  143|    775|        const ptrdiff_t stride = out->stride[1];
  144|    775|        const ptrdiff_t sz = ((out->p.h + ss_ver) >> ss_ver) * stride;
  145|    775|        if (sz < 0) {
  ------------------
  |  Branch (145:13): [True: 0, False: 775]
  ------------------
  146|      0|            if (!data->num_uv_points[0])
  ------------------
  |  Branch (146:17): [True: 0, False: 0]
  ------------------
  147|      0|                memcpy((uint8_t*) out->data[1] + sz - stride,
  148|      0|                       (uint8_t*) in->data[1] + sz - stride, -sz);
  149|      0|            if (!data->num_uv_points[1])
  ------------------
  |  Branch (149:17): [True: 0, False: 0]
  ------------------
  150|      0|                memcpy((uint8_t*) out->data[2] + sz - stride,
  151|      0|                       (uint8_t*) in->data[2] + sz - stride, -sz);
  152|    775|        } else {
  153|    775|            if (!data->num_uv_points[0])
  ------------------
  |  Branch (153:17): [True: 349, False: 426]
  ------------------
  154|    349|                memcpy(out->data[1], in->data[1], sz);
  155|    775|            if (!data->num_uv_points[1])
  ------------------
  |  Branch (155:17): [True: 512, False: 263]
  ------------------
  156|    512|                memcpy(out->data[2], in->data[2], sz);
  157|    775|        }
  158|    775|    }
  159|  2.40k|}
dav1d_apply_grain_row_16bpc:
  167|  9.15k|{
  168|       |    // Synthesize grain for the affected planes
  169|  9.15k|    const Dav1dFilmGrainData *const data = &out->frame_hdr->film_grain.data;
  170|  9.15k|    const int ss_y = in->p.layout == DAV1D_PIXEL_LAYOUT_I420;
  171|  9.15k|    const int ss_x = in->p.layout != DAV1D_PIXEL_LAYOUT_I444;
  172|  9.15k|    const int cpw = (out->p.w + ss_x) >> ss_x;
  173|  9.15k|    const int is_id = out->seq_hdr->mtrx == DAV1D_MC_IDENTITY;
  174|  9.15k|    pixel *const luma_src =
  175|  9.15k|        ((pixel *) in->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(in->stride[0]);
  ------------------
  |  |   37|  9.15k|#define FG_BLOCK_SIZE 32
  ------------------
  176|  9.15k|#if BITDEPTH != 8
  177|  9.15k|    const int bitdepth_max = (1 << out->p.bpc) - 1;
  178|  9.15k|#endif
  179|       |
  180|  9.15k|    if (data->num_y_points) {
  ------------------
  |  Branch (180:9): [True: 3.33k, False: 5.82k]
  ------------------
  181|  3.33k|        const int bh = imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE);
  ------------------
  |  |   37|  3.33k|#define FG_BLOCK_SIZE 32
  ------------------
                      const int bh = imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE);
  ------------------
  |  |   37|  3.33k|#define FG_BLOCK_SIZE 32
  ------------------
  182|  3.33k|        dsp->fgy_32x32xn(((pixel *) out->data[0]) + row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[0]),
  ------------------
  |  |   37|  3.33k|#define FG_BLOCK_SIZE 32
  ------------------
  183|  3.33k|                         luma_src, out->stride[0], data,
  184|  3.33k|                         out->p.w, scaling[0], grain_lut[0], bh, row HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  3.33k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  185|  3.33k|    }
  186|       |
  187|  9.15k|    if (!data->num_uv_points[0] && !data->num_uv_points[1] &&
  ------------------
  |  Branch (187:9): [True: 8.03k, False: 1.12k]
  |  Branch (187:36): [True: 7.36k, False: 668]
  ------------------
  188|  7.37k|        !data->chroma_scaling_from_luma)
  ------------------
  |  Branch (188:9): [True: 1.73k, False: 5.64k]
  ------------------
  189|  1.73k|    {
  190|  1.73k|        return;
  191|  1.73k|    }
  192|       |
  193|  7.42k|    const int bh = (imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE) + ss_y) >> ss_y;
  ------------------
  |  |   37|  7.42k|#define FG_BLOCK_SIZE 32
  ------------------
                  const int bh = (imin(out->p.h - row * FG_BLOCK_SIZE, FG_BLOCK_SIZE) + ss_y) >> ss_y;
  ------------------
  |  |   37|  7.42k|#define FG_BLOCK_SIZE 32
  ------------------
  194|       |
  195|       |    // extend padding pixels
  196|  7.42k|    if (out->p.w & ss_x) {
  ------------------
  |  Branch (196:9): [True: 2.57k, False: 4.85k]
  ------------------
  197|  2.57k|        pixel *ptr = luma_src;
  198|  36.6k|        for (int y = 0; y < bh; y++) {
  ------------------
  |  Branch (198:25): [True: 34.1k, False: 2.57k]
  ------------------
  199|  34.1k|            ptr[out->p.w] = ptr[out->p.w - 1];
  200|  34.1k|            ptr += PXSTRIDE(in->stride[0]) << ss_y;
  201|  34.1k|        }
  202|  2.57k|    }
  203|       |
  204|  7.42k|    const ptrdiff_t uv_off = row * FG_BLOCK_SIZE * PXSTRIDE(out->stride[1]) >> ss_y;
  ------------------
  |  |   37|  7.42k|#define FG_BLOCK_SIZE 32
  ------------------
  205|  7.42k|    if (data->chroma_scaling_from_luma) {
  ------------------
  |  Branch (205:9): [True: 5.61k, False: 1.81k]
  ------------------
  206|  16.7k|        for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (206:26): [True: 11.1k, False: 5.61k]
  ------------------
  207|  11.1k|            dsp->fguv_32x32xn[in->p.layout - 1](((pixel *) out->data[1 + pl]) + uv_off,
  208|  11.1k|                                                ((const pixel *) in->data[1 + pl]) + uv_off,
  209|  11.1k|                                                in->stride[1], data, cpw,
  210|  11.1k|                                                scaling[0], grain_lut[1 + pl],
  211|  11.1k|                                                bh, row, luma_src, in->stride[0],
  212|  11.1k|                                                pl, is_id HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  11.1k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  213|  5.61k|    } else {
  214|  5.30k|        for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (214:26): [True: 3.48k, False: 1.81k]
  ------------------
  215|  3.48k|            if (data->num_uv_points[pl])
  ------------------
  |  Branch (215:17): [True: 2.20k, False: 1.28k]
  ------------------
  216|  2.20k|                dsp->fguv_32x32xn[in->p.layout - 1](((pixel *) out->data[1 + pl]) + uv_off,
  217|  2.20k|                                                    ((const pixel *) in->data[1 + pl]) + uv_off,
  218|  2.20k|                                                    in->stride[1], data, cpw,
  219|  2.20k|                                                    scaling[1 + pl], grain_lut[1 + pl],
  220|  2.20k|                                                    bh, row, luma_src, in->stride[0],
  221|  2.20k|                                                    pl, is_id HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |   74|  2.20k|#define HIGHBD_TAIL_SUFFIX , bitdepth_max
  ------------------
  222|  1.81k|    }
  223|  7.42k|}

dav1d_film_grain_dsp_init_8bpc:
  425|  3.46k|COLD void bitfn(dav1d_film_grain_dsp_init)(Dav1dFilmGrainDSPContext *const c) {
  426|  3.46k|    c->generate_grain_y = generate_grain_y_c;
  427|  3.46k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I420 - 1] = generate_grain_uv_420_c;
  428|  3.46k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I422 - 1] = generate_grain_uv_422_c;
  429|  3.46k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I444 - 1] = generate_grain_uv_444_c;
  430|       |
  431|  3.46k|    c->fgy_32x32xn = fgy_32x32xn_c;
  432|  3.46k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = fguv_32x32xn_420_c;
  433|  3.46k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = fguv_32x32xn_422_c;
  434|  3.46k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = fguv_32x32xn_444_c;
  435|       |
  436|  3.46k|#if HAVE_ASM
  437|       |#if ARCH_AARCH64 || ARCH_ARM
  438|       |    film_grain_dsp_init_arm(c);
  439|       |#elif ARCH_X86
  440|       |    film_grain_dsp_init_x86(c);
  441|       |#elif ARCH_RISCV
  442|       |    film_grain_dsp_init_riscv(c);
  443|       |#endif
  444|  3.46k|#endif
  445|  3.46k|}
dav1d_film_grain_dsp_init_16bpc:
  425|  5.11k|COLD void bitfn(dav1d_film_grain_dsp_init)(Dav1dFilmGrainDSPContext *const c) {
  426|  5.11k|    c->generate_grain_y = generate_grain_y_c;
  427|  5.11k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I420 - 1] = generate_grain_uv_420_c;
  428|  5.11k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I422 - 1] = generate_grain_uv_422_c;
  429|  5.11k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I444 - 1] = generate_grain_uv_444_c;
  430|       |
  431|  5.11k|    c->fgy_32x32xn = fgy_32x32xn_c;
  432|  5.11k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = fguv_32x32xn_420_c;
  433|  5.11k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = fguv_32x32xn_422_c;
  434|  5.11k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = fguv_32x32xn_444_c;
  435|       |
  436|  5.11k|#if HAVE_ASM
  437|       |#if ARCH_AARCH64 || ARCH_ARM
  438|       |    film_grain_dsp_init_arm(c);
  439|       |#elif ARCH_X86
  440|       |    film_grain_dsp_init_x86(c);
  441|       |#elif ARCH_RISCV
  442|       |    film_grain_dsp_init_riscv(c);
  443|       |#endif
  444|  5.11k|#endif
  445|  5.11k|}

dav1d_init_get_bits:
   38|   549k|{
   39|   549k|    assert(sz);
  ------------------
  |  Branch (39:5): [True: 549k, False: 0]
  ------------------
   40|   549k|    c->ptr = c->ptr_start = data;
   41|   549k|    c->ptr_end = &c->ptr_start[sz];
   42|   549k|    c->state = 0;
   43|   549k|    c->bits_left = 0;
   44|   549k|    c->error = 0;
   45|   549k|}
dav1d_get_bit:
   47|  12.3M|unsigned dav1d_get_bit(GetBits *const c) {
   48|  12.3M|    if (!c->bits_left) {
  ------------------
  |  Branch (48:9): [True: 2.03M, False: 10.2M]
  ------------------
   49|  2.03M|        if (c->ptr >= c->ptr_end) {
  ------------------
  |  Branch (49:13): [True: 11.7k, False: 2.01M]
  ------------------
   50|  11.7k|            c->error = 1;
   51|  2.01M|        } else {
   52|  2.01M|            const unsigned state = *c->ptr++;
   53|  2.01M|            c->bits_left = 7;
   54|  2.01M|            c->state = (uint64_t) state << 57;
   55|  2.01M|            return state >> 7;
   56|  2.01M|        }
   57|  2.03M|    }
   58|       |
   59|  10.3M|    const uint64_t state = c->state;
   60|  10.3M|    c->bits_left--;
   61|  10.3M|    c->state = state << 1;
   62|  10.3M|    return (unsigned) (state >> 63);
   63|  12.3M|}
dav1d_get_uleb128:
   95|   187k|unsigned dav1d_get_uleb128(GetBits *const c) {
   96|   187k|    uint64_t val = 0;
   97|   187k|    unsigned i = 0, more;
   98|       |
   99|   197k|    do {
  100|   197k|        const int v = dav1d_get_bits(c, 8);
  101|   197k|        more = v & 0x80;
  102|   197k|        val |= ((uint64_t) (v & 0x7F)) << i;
  103|   197k|        i += 7;
  104|   197k|    } while (more && i < 56);
  ------------------
  |  Branch (104:14): [True: 9.86k, False: 187k]
  |  Branch (104:22): [True: 9.77k, False: 88]
  ------------------
  105|       |
  106|   187k|    if (val > UINT32_MAX || more) {
  ------------------
  |  Branch (106:9): [True: 669, False: 187k]
  |  Branch (106:29): [True: 55, False: 187k]
  ------------------
  107|    724|        c->error = 1;
  108|    724|        return 0;
  109|    724|    }
  110|       |
  111|   187k|    return (unsigned) val;
  112|   187k|}
dav1d_get_uniform:
  114|   178k|unsigned dav1d_get_uniform(GetBits *const c, const unsigned max) {
  115|       |    // Output in range [0..max-1]
  116|       |    // max must be > 1, or else nothing is read from the bitstream
  117|   178k|    assert(max > 1);
  ------------------
  |  Branch (117:5): [True: 178k, False: 0]
  ------------------
  118|   178k|    const int l = ulog2(max) + 1;
  119|   178k|    assert(l > 1);
  ------------------
  |  Branch (119:5): [True: 178k, False: 0]
  ------------------
  120|   178k|    const unsigned m = (1U << l) - max;
  121|   178k|    const unsigned v = dav1d_get_bits(c, l - 1);
  122|   178k|    return v < m ? v : (v << 1) - m + dav1d_get_bit(c);
  ------------------
  |  Branch (122:12): [True: 165k, False: 12.8k]
  ------------------
  123|   178k|}
dav1d_get_vlc:
  125|    650|unsigned dav1d_get_vlc(GetBits *const c) {
  126|    650|    if (dav1d_get_bit(c))
  ------------------
  |  Branch (126:9): [True: 260, False: 390]
  ------------------
  127|    260|        return 0;
  128|       |
  129|    390|    int n_bits = 0;
  130|  7.09k|    do {
  131|  7.09k|        if (++n_bits == 32)
  ------------------
  |  Branch (131:13): [True: 73, False: 7.02k]
  ------------------
  132|     73|            return UINT32_MAX;
  133|  7.09k|    } while (!dav1d_get_bit(c));
  ------------------
  |  Branch (133:14): [True: 6.70k, False: 317]
  ------------------
  134|       |
  135|    317|    return ((1U << n_bits) - 1) + dav1d_get_bits(c, n_bits);
  136|    390|}
dav1d_get_bits_subexp:
  162|   178k|int dav1d_get_bits_subexp(GetBits *const c, const int ref, const unsigned n) {
  163|   178k|    return (int) get_bits_subexp_u(c, ref + (1 << n), 2 << n) - (1 << n);
  164|   178k|}
getbits.c:refill:
   65|  2.10M|static inline void refill(GetBits *const c, const int n) {
   66|  2.10M|    assert(c->bits_left >= 0 && c->bits_left < 32);
  ------------------
  |  Branch (66:5): [True: 2.10M, False: 0]
  |  Branch (66:5): [True: 2.10M, False: 0]
  ------------------
   67|  2.10M|    unsigned state = 0;
   68|  2.20M|    do {
   69|  2.20M|        if (c->ptr >= c->ptr_end) {
  ------------------
  |  Branch (69:13): [True: 16.3k, False: 2.19M]
  ------------------
   70|  16.3k|            c->error = 1;
   71|  16.3k|            if (state) break;
  ------------------
  |  Branch (71:17): [True: 1.04k, False: 15.3k]
  ------------------
   72|  15.3k|            return;
   73|  16.3k|        }
   74|  2.19M|        state = (state << 8) | *c->ptr++;
   75|  2.19M|        c->bits_left += 8;
   76|  2.19M|    } while (n > c->bits_left);
  ------------------
  |  Branch (76:14): [True: 99.5k, False: 2.09M]
  ------------------
   77|  2.09M|    c->state |= (uint64_t) state << (64 - c->bits_left);
   78|  2.09M|}
getbits.c:get_bits_subexp_u:
  140|   178k|{
  141|   178k|    unsigned v = 0;
  142|       |
  143|   337k|    for (int i = 0;; i++) {
  144|   337k|        const int b = i ? 3 + i - 1 : 3;
  ------------------
  |  Branch (144:23): [True: 158k, False: 178k]
  ------------------
  145|       |
  146|   337k|        if (n < v + 3 * (1 << b)) {
  ------------------
  |  Branch (146:13): [True: 8.09k, False: 329k]
  ------------------
  147|  8.09k|            v += dav1d_get_uniform(c, n - v + 1);
  148|  8.09k|            break;
  149|  8.09k|        }
  150|       |
  151|   329k|        if (!dav1d_get_bit(c)) {
  ------------------
  |  Branch (151:13): [True: 170k, False: 158k]
  ------------------
  152|   170k|            v += dav1d_get_bits(c, b);
  153|   170k|            break;
  154|   170k|        }
  155|       |
  156|   158k|        v += 1 << b;
  157|   158k|    }
  158|       |
  159|   178k|    return ref * 2 <= n ? inv_recenter(ref, v) : n - inv_recenter(n - ref, v);
  ------------------
  |  Branch (159:12): [True: 168k, False: 10.1k]
  ------------------
  160|   178k|}

obu.c:dav1d_bytealign_get_bits:
   52|   756k|static inline void dav1d_bytealign_get_bits(GetBits *c) {
   53|       |    // bits_left is never more than 7, because it is only incremented
   54|       |    // by refill(), called by dav1d_get_bits and that never reads more
   55|       |    // than 7 bits more than it needs.
   56|       |    //
   57|       |    // If this wasn't true, we would need to work out how many bits to
   58|       |    // discard (bits_left % 8), subtract that from bits_left and then
   59|       |    // shift state right by that amount.
   60|   756k|    assert(c->bits_left <= 7);
  ------------------
  |  Branch (60:5): [True: 756k, False: 0]
  ------------------
   61|       |
   62|   756k|    c->bits_left = 0;
   63|   756k|    c->state = 0;
   64|   756k|}

dav1d_init_intra_edge_tree:
  126|      1|COLD void dav1d_init_intra_edge_tree(void) {
  127|       |    // This function is guaranteed to be called only once
  128|      1|    struct ModeSelMem mem;
  129|       |
  130|      1|    mem.nwc[BL_128X128] = &nodes.branch_sb128[1];
  131|      1|    mem.nwc[BL_64X64] = &nodes.branch_sb128[1 + 4];
  132|      1|    mem.nwc[BL_32X32] = &nodes.branch_sb128[1 + 4 + 16];
  133|      1|    mem.nt = nodes.tip_sb128;
  134|      1|    init_mode_node(nodes.branch_sb128, BL_128X128, &mem, 1, 0);
  135|      1|    assert(mem.nwc[BL_128X128] == &nodes.branch_sb128[1 + 4]);
  ------------------
  |  Branch (135:5): [True: 1, False: 0]
  ------------------
  136|      1|    assert(mem.nwc[BL_64X64] == &nodes.branch_sb128[1 + 4 + 16]);
  ------------------
  |  Branch (136:5): [True: 1, False: 0]
  ------------------
  137|      1|    assert(mem.nwc[BL_32X32] == &nodes.branch_sb128[1 + 4 + 16 + 64]);
  ------------------
  |  Branch (137:5): [True: 1, False: 0]
  ------------------
  138|      1|    assert(mem.nt == &nodes.tip_sb128[256]);
  ------------------
  |  Branch (138:5): [True: 1, False: 0]
  ------------------
  139|       |
  140|      1|    mem.nwc[BL_128X128] = NULL;
  141|      1|    mem.nwc[BL_64X64] = &nodes.branch_sb64[1];
  142|      1|    mem.nwc[BL_32X32] = &nodes.branch_sb64[1 + 4];
  143|      1|    mem.nt = nodes.tip_sb64;
  144|      1|    init_mode_node(nodes.branch_sb64, BL_64X64, &mem, 1, 0);
  145|      1|    assert(mem.nwc[BL_64X64] == &nodes.branch_sb64[1 + 4]);
  ------------------
  |  Branch (145:5): [True: 1, False: 0]
  ------------------
  146|      1|    assert(mem.nwc[BL_32X32] == &nodes.branch_sb64[1 + 4 + 16]);
  ------------------
  |  Branch (146:5): [True: 1, False: 0]
  ------------------
  147|      1|    assert(mem.nt == &nodes.tip_sb64[64]);
  ------------------
  |  Branch (147:5): [True: 1, False: 0]
  ------------------
  148|      1|}
intra_edge.c:init_mode_node:
  101|    106|{
  102|    106|    init_edges(&nwc->node, bl,
  103|    106|               (top_has_right ? EDGE_ALL_TOP_HAS_RIGHT : 0) |
  ------------------
  |  Branch (103:17): [True: 73, False: 33]
  ------------------
  104|    106|               (left_has_bottom ? EDGE_ALL_LEFT_HAS_BOTTOM : 0));
  ------------------
  |  Branch (104:17): [True: 33, False: 73]
  ------------------
  105|    106|    if (bl == BL_16X16) {
  ------------------
  |  Branch (105:9): [True: 80, False: 26]
  ------------------
  106|    400|        for (int n = 0; n < 4; n++) {
  ------------------
  |  Branch (106:25): [True: 320, False: 80]
  ------------------
  107|    320|            EdgeTip *const nt = mem->nt++;
  108|    320|            nwc->split_offset[n] = PTR_OFFSET(nwc, nt);
  ------------------
  |  |   94|    320|#define PTR_OFFSET(a, b) ((uint16_t)((uintptr_t)(b) - (uintptr_t)(a)))
  ------------------
  109|    320|            init_edges(&nt->node, bl + 1,
  110|    320|                       ((n == 3 || (n == 1 && !top_has_right)) ? 0 :
  ------------------
  |  Branch (110:26): [True: 80, False: 240]
  |  Branch (110:37): [True: 80, False: 160]
  |  Branch (110:47): [True: 26, False: 54]
  ------------------
  111|    320|                        EDGE_ALL_TOP_HAS_RIGHT) |
  112|    320|                       (!(n == 0 || (n == 2 && left_has_bottom)) ? 0 :
  ------------------
  |  Branch (112:27): [True: 80, False: 240]
  |  Branch (112:38): [True: 80, False: 160]
  |  Branch (112:48): [True: 26, False: 54]
  ------------------
  113|    320|                        EDGE_ALL_LEFT_HAS_BOTTOM));
  114|    320|        }
  115|     80|    } else {
  116|    130|        for (int n = 0; n < 4; n++) {
  ------------------
  |  Branch (116:25): [True: 104, False: 26]
  ------------------
  117|    104|            EdgeBranch *const nwc_child = mem->nwc[bl]++;
  118|    104|            nwc->split_offset[n] = PTR_OFFSET(nwc, nwc_child);
  ------------------
  |  |   94|    104|#define PTR_OFFSET(a, b) ((uint16_t)((uintptr_t)(b) - (uintptr_t)(a)))
  ------------------
  119|    104|            init_mode_node(nwc_child, bl + 1, mem,
  120|    104|                           !(n == 3 || (n == 1 && !top_has_right)),
  ------------------
  |  Branch (120:30): [True: 26, False: 78]
  |  Branch (120:41): [True: 26, False: 52]
  |  Branch (120:51): [True: 7, False: 19]
  ------------------
  121|    104|                           n == 0 || (n == 2 && left_has_bottom));
  ------------------
  |  Branch (121:28): [True: 26, False: 78]
  |  Branch (121:39): [True: 26, False: 52]
  |  Branch (121:49): [True: 7, False: 19]
  ------------------
  122|    104|        }
  123|     26|    }
  124|    106|}
intra_edge.c:init_edges:
   58|    426|{
   59|    426|    node->o = edge_flags;
   60|    426|    node->h[0] = edge_flags | EDGE_ALL_LEFT_HAS_BOTTOM;
   61|    426|    node->v[0] = edge_flags | EDGE_ALL_TOP_HAS_RIGHT;
   62|       |
   63|    426|    if (bl == BL_8X8) {
  ------------------
  |  Branch (63:9): [True: 320, False: 106]
  ------------------
   64|    320|        EdgeTip *const nt = (EdgeTip *) node;
   65|       |
   66|    320|        node->h[1] = edge_flags & (EDGE_ALL_LEFT_HAS_BOTTOM |
   67|    320|                                   EDGE_I420_TOP_HAS_RIGHT);
   68|    320|        node->v[1] = edge_flags & (EDGE_ALL_TOP_HAS_RIGHT |
   69|    320|                                   EDGE_I420_LEFT_HAS_BOTTOM |
   70|    320|                                   EDGE_I422_LEFT_HAS_BOTTOM);
   71|       |
   72|    320|        nt->split[0] = (edge_flags & EDGE_ALL_TOP_HAS_RIGHT) |
   73|    320|                       EDGE_I422_LEFT_HAS_BOTTOM;
   74|    320|        nt->split[1] = edge_flags | EDGE_I444_TOP_HAS_RIGHT;
   75|    320|        nt->split[2] = edge_flags & (EDGE_I420_TOP_HAS_RIGHT |
   76|    320|                                     EDGE_I420_LEFT_HAS_BOTTOM |
   77|    320|                                     EDGE_I422_LEFT_HAS_BOTTOM);
   78|    320|    } else {
   79|    106|        EdgeBranch *const nwc = (EdgeBranch *) node;
   80|       |
   81|    106|        node->h[1] = edge_flags & EDGE_ALL_LEFT_HAS_BOTTOM;
   82|    106|        node->v[1] = edge_flags & EDGE_ALL_TOP_HAS_RIGHT;
   83|       |
   84|    106|        nwc->h4 = EDGE_ALL_LEFT_HAS_BOTTOM;
   85|    106|        nwc->v4 = EDGE_ALL_TOP_HAS_RIGHT;
   86|    106|        if (bl == BL_16X16) {
  ------------------
  |  Branch (86:13): [True: 80, False: 26]
  ------------------
   87|     80|            nwc->h4 |= edge_flags & EDGE_I420_TOP_HAS_RIGHT;
   88|     80|            nwc->v4 |= edge_flags & (EDGE_I420_LEFT_HAS_BOTTOM |
   89|     80|                                     EDGE_I422_LEFT_HAS_BOTTOM);
   90|     80|        }
   91|    106|    }
   92|    426|}

recon_tmpl.c:sm_flag:
   95|  4.75M|static inline int sm_flag(const BlockContext *const b, const int idx) {
   96|  4.75M|    if (!b->intra[idx]) return 0;
  ------------------
  |  Branch (96:9): [True: 249k, False: 4.50M]
  ------------------
   97|  4.50M|    const enum IntraPredMode m = b->mode[idx];
   98|  4.50M|    return (m == SMOOTH_PRED || m == SMOOTH_H_PRED ||
  ------------------
  |  Branch (98:13): [True: 165k, False: 4.33M]
  |  Branch (98:33): [True: 78.9k, False: 4.26M]
  ------------------
   99|  4.26M|            m == SMOOTH_V_PRED) ? ANGLE_SMOOTH_EDGE_FLAG : 0;
  ------------------
  |  |   93|   305k|#define ANGLE_SMOOTH_EDGE_FLAG      512
  ------------------
  |  Branch (99:13): [True: 57.7k, False: 4.20M]
  ------------------
  100|  4.75M|}
recon_tmpl.c:sm_uv_flag:
  102|  2.65M|static inline int sm_uv_flag(const BlockContext *const b, const int idx) {
  103|  2.65M|    const enum IntraPredMode m = b->uvmode[idx];
  104|  2.65M|    return (m == SMOOTH_PRED || m == SMOOTH_H_PRED ||
  ------------------
  |  Branch (104:13): [True: 90.3k, False: 2.56M]
  |  Branch (104:33): [True: 57.1k, False: 2.51M]
  ------------------
  105|  2.51M|            m == SMOOTH_V_PRED) ? ANGLE_SMOOTH_EDGE_FLAG : 0;
  ------------------
  |  |   93|   188k|#define ANGLE_SMOOTH_EDGE_FLAG      512
  ------------------
  |  Branch (105:13): [True: 40.0k, False: 2.47M]
  ------------------
  106|  2.65M|}

dav1d_prepare_intra_edges_8bpc:
   86|  13.6M|{
   87|  13.6M|    const int bitdepth = bitdepth_from_max(bitdepth_max);
  ------------------
  |  |   58|  13.6M|#define bitdepth_from_max(x) 8
  ------------------
   88|  13.6M|    assert(y < h && x < w);
  ------------------
  |  Branch (88:5): [True: 13.6M, False: 4.12k]
  |  Branch (88:5): [True: 13.6M, False: 18.4E]
  ------------------
   89|       |
   90|  13.6M|    switch (mode) {
   91|   203k|    case VERT_PRED:
  ------------------
  |  Branch (91:5): [True: 203k, False: 13.4M]
  ------------------
   92|   554k|    case HOR_PRED:
  ------------------
  |  Branch (92:5): [True: 351k, False: 13.3M]
  ------------------
   93|   603k|    case DIAG_DOWN_LEFT_PRED:
  ------------------
  |  Branch (93:5): [True: 48.3k, False: 13.6M]
  ------------------
   94|   652k|    case DIAG_DOWN_RIGHT_PRED:
  ------------------
  |  Branch (94:5): [True: 49.4k, False: 13.6M]
  ------------------
   95|   699k|    case VERT_RIGHT_PRED:
  ------------------
  |  Branch (95:5): [True: 46.5k, False: 13.6M]
  ------------------
   96|  1.57M|    case HOR_DOWN_PRED:
  ------------------
  |  Branch (96:5): [True: 874k, False: 12.7M]
  ------------------
   97|  1.72M|    case HOR_UP_PRED:
  ------------------
  |  Branch (97:5): [True: 147k, False: 13.5M]
  ------------------
   98|  1.77M|    case VERT_LEFT_PRED: {
  ------------------
  |  Branch (98:5): [True: 57.4k, False: 13.6M]
  ------------------
   99|  1.77M|        *angle = av1_mode_to_angle_map[mode - VERT_PRED] + 3 * *angle;
  100|       |
  101|  1.77M|        if (*angle <= 90)
  ------------------
  |  Branch (101:13): [True: 260k, False: 1.51M]
  ------------------
  102|   260k|            mode = *angle < 90 && have_top ? Z1_PRED : VERT_PRED;
  ------------------
  |  Branch (102:20): [True: 144k, False: 116k]
  |  Branch (102:35): [True: 142k, False: 1.79k]
  ------------------
  103|  1.51M|        else if (*angle < 180)
  ------------------
  |  Branch (103:18): [True: 1.06M, False: 452k]
  ------------------
  104|  1.06M|            mode = Z2_PRED;
  105|   452k|        else
  106|   452k|            mode = *angle > 180 && have_left ? Z3_PRED : HOR_PRED;
  ------------------
  |  Branch (106:20): [True: 216k, False: 236k]
  |  Branch (106:36): [True: 211k, False: 4.48k]
  ------------------
  107|  1.77M|        break;
  108|  1.72M|    }
  109|  10.9M|    case DC_PRED:
  ------------------
  |  Branch (109:5): [True: 10.9M, False: 2.73M]
  ------------------
  110|  11.2M|    case PAETH_PRED:
  ------------------
  |  Branch (110:5): [True: 297k, False: 13.3M]
  ------------------
  111|  11.2M|        mode = av1_mode_conv[mode][have_left][have_top];
  112|  11.2M|        break;
  113|   672k|    default:
  ------------------
  |  Branch (113:5): [True: 672k, False: 12.9M]
  ------------------
  114|   672k|        break;
  115|  13.6M|    }
  116|       |
  117|  13.6M|    const pixel *dst_top;
  118|  13.6M|    if (have_top &&
  ------------------
  |  Branch (118:9): [True: 13.5M, False: 167k]
  ------------------
  119|  13.5M|        (av1_intra_prediction_edges[mode].needs_top ||
  ------------------
  |  Branch (119:10): [True: 13.0M, False: 440k]
  ------------------
  120|   440k|         av1_intra_prediction_edges[mode].needs_topleft ||
  ------------------
  |  Branch (120:10): [True: 205k, False: 235k]
  ------------------
  121|   235k|         (av1_intra_prediction_edges[mode].needs_left && !have_left)))
  ------------------
  |  Branch (121:11): [True: 235k, False: 0]
  |  Branch (121:58): [True: 9.31k, False: 225k]
  ------------------
  122|  13.2M|    {
  123|  13.2M|        if (prefilter_toplevel_sb_edge) {
  ------------------
  |  Branch (123:13): [True: 1.04M, False: 12.2M]
  ------------------
  124|  1.04M|            dst_top = &prefilter_toplevel_sb_edge[x * 4];
  125|  12.2M|        } else {
  126|  12.2M|            dst_top = &dst[-PXSTRIDE(stride)];
  ------------------
  |  |   53|  12.2M|#define PXSTRIDE(x) (x)
  ------------------
  127|  12.2M|        }
  128|  13.2M|    }
  129|       |
  130|  13.6M|    if (av1_intra_prediction_edges[mode].needs_left) {
  ------------------
  |  Branch (130:9): [True: 12.3M, False: 1.35M]
  ------------------
  131|  12.3M|        const int sz = th << 2;
  132|  12.3M|        pixel *const left = &topleft_out[-sz];
  133|       |
  134|  12.3M|        if (have_left) {
  ------------------
  |  Branch (134:13): [True: 12.2M, False: 116k]
  ------------------
  135|  12.2M|            const int px_have = imin(sz, (h - y) << 2);
  136|       |
  137|  95.9M|            for (int i = 0; i < px_have; i++)
  ------------------
  |  Branch (137:29): [True: 83.7M, False: 12.2M]
  ------------------
  138|  83.7M|                left[sz - 1 - i] = dst[PXSTRIDE(stride) * i - 1];
  ------------------
  |  |   53|  83.7M|#define PXSTRIDE(x) (x)
  ------------------
  139|  12.2M|            if (px_have < sz)
  ------------------
  |  Branch (139:17): [True: 21.8k, False: 12.1M]
  ------------------
  140|  21.8k|                pixel_set(left, left[sz - px_have], sz - px_have);
  ------------------
  |  |   48|  21.8k|#define pixel_set memset
  ------------------
  141|  12.2M|        } else {
  142|   116k|            pixel_set(left, have_top ? *dst_top : ((1 << bitdepth) >> 1) + 1, sz);
  ------------------
  |  |   48|   116k|#define pixel_set memset
  ------------------
  |  Branch (142:29): [True: 111k, False: 4.57k]
  ------------------
  143|   116k|        }
  144|       |
  145|  12.3M|        if (av1_intra_prediction_edges[mode].needs_bottomleft) {
  ------------------
  |  Branch (145:13): [True: 211k, False: 12.1M]
  ------------------
  146|   211k|            const int have_bottomleft = (!have_left || y + th >= h) ? 0 :
  ------------------
  |  Branch (146:42): [True: 18.4E, False: 211k]
  |  Branch (146:56): [True: 4.36k, False: 207k]
  ------------------
  147|   211k|                                        (edge_flags & EDGE_I444_LEFT_HAS_BOTTOM);
  148|       |
  149|   211k|            if (have_bottomleft) {
  ------------------
  |  Branch (149:17): [True: 45.6k, False: 165k]
  ------------------
  150|  45.6k|                const int px_have = imin(sz, (h - y - th) << 2);
  151|       |
  152|   417k|                for (int i = 0; i < px_have; i++)
  ------------------
  |  Branch (152:33): [True: 371k, False: 45.6k]
  ------------------
  153|   371k|                    left[-(i + 1)] = dst[(sz + i) * PXSTRIDE(stride) - 1];
  ------------------
  |  |   53|   371k|#define PXSTRIDE(x) (x)
  ------------------
  154|  45.6k|                if (px_have < sz)
  ------------------
  |  Branch (154:21): [True: 418, False: 45.2k]
  ------------------
  155|    418|                    pixel_set(left - sz, left[-px_have], sz - px_have);
  ------------------
  |  |   48|    418|#define pixel_set memset
  ------------------
  156|   165k|            } else {
  157|   165k|                pixel_set(left - sz, left[0], sz);
  ------------------
  |  |   48|   165k|#define pixel_set memset
  ------------------
  158|   165k|            }
  159|   211k|        }
  160|  12.3M|    }
  161|       |
  162|  13.6M|    if (av1_intra_prediction_edges[mode].needs_top) {
  ------------------
  |  Branch (162:9): [True: 13.1M, False: 560k]
  ------------------
  163|  13.1M|        const int sz = tw << 2;
  164|  13.1M|        pixel *const top = &topleft_out[1];
  165|       |
  166|  13.1M|        if (have_top) {
  ------------------
  |  Branch (166:13): [True: 13.0M, False: 44.9k]
  ------------------
  167|  13.0M|            const int px_have = imin(sz, (w - x) << 2);
  168|  13.0M|            pixel_copy(top, dst_top, px_have);
  ------------------
  |  |   47|  13.0M|#define pixel_copy memcpy
  ------------------
  169|  13.0M|            if (px_have < sz)
  ------------------
  |  Branch (169:17): [True: 416k, False: 12.6M]
  ------------------
  170|   416k|                pixel_set(top + px_have, top[px_have - 1], sz - px_have);
  ------------------
  |  |   48|   416k|#define pixel_set memset
  ------------------
  171|  13.0M|        } else {
  172|  44.9k|            pixel_set(top, have_left ? dst[-1] : ((1 << bitdepth) >> 1) - 1, sz);
  ------------------
  |  |   48|  44.9k|#define pixel_set memset
  ------------------
  |  Branch (172:28): [True: 43.8k, False: 1.15k]
  ------------------
  173|  44.9k|        }
  174|       |
  175|  13.1M|        if (av1_intra_prediction_edges[mode].needs_topright) {
  ------------------
  |  Branch (175:13): [True: 142k, False: 12.9M]
  ------------------
  176|   142k|            const int have_topright = (!have_top || x + tw >= w) ? 0 :
  ------------------
  |  Branch (176:40): [True: 18.4E, False: 142k]
  |  Branch (176:53): [True: 5.42k, False: 137k]
  ------------------
  177|   142k|                                      (edge_flags & EDGE_I444_TOP_HAS_RIGHT);
  178|       |
  179|   142k|            if (have_topright) {
  ------------------
  |  Branch (179:17): [True: 100k, False: 41.6k]
  ------------------
  180|   100k|                const int px_have = imin(sz, (w - x - tw) << 2);
  181|       |
  182|   100k|                pixel_copy(top + sz, &dst_top[sz], px_have);
  ------------------
  |  |   47|   100k|#define pixel_copy memcpy
  ------------------
  183|   100k|                if (px_have < sz)
  ------------------
  |  Branch (183:21): [True: 1.52k, False: 99.2k]
  ------------------
  184|  1.52k|                    pixel_set(top + sz + px_have, top[sz + px_have - 1],
  ------------------
  |  |   48|  1.52k|#define pixel_set memset
  ------------------
  185|  1.52k|                              sz - px_have);
  186|   100k|            } else {
  187|  41.6k|                pixel_set(top + sz, top[sz - 1], sz);
  ------------------
  |  |   48|  41.6k|#define pixel_set memset
  ------------------
  188|  41.6k|            }
  189|   142k|        }
  190|  13.1M|    }
  191|       |
  192|  13.6M|    if (av1_intra_prediction_edges[mode].needs_topleft) {
  ------------------
  |  Branch (192:9): [True: 1.91M, False: 11.7M]
  ------------------
  193|  1.91M|        if (have_left)
  ------------------
  |  Branch (193:13): [True: 1.81M, False: 96.3k]
  ------------------
  194|  1.81M|            *topleft_out = have_top ? dst_top[-1] : dst[-1];
  ------------------
  |  Branch (194:28): [True: 1.77M, False: 35.9k]
  ------------------
  195|  96.3k|        else
  196|  96.3k|            *topleft_out = have_top ? *dst_top : (1 << bitdepth) >> 1;
  ------------------
  |  Branch (196:28): [True: 93.2k, False: 3.09k]
  ------------------
  197|       |
  198|  1.91M|        if (mode == Z2_PRED && tw + th >= 6 && filter_edge)
  ------------------
  |  Branch (198:13): [True: 1.06M, False: 846k]
  |  Branch (198:32): [True: 49.5k, False: 1.01M]
  |  Branch (198:48): [True: 45.3k, False: 4.21k]
  ------------------
  199|  45.3k|            *topleft_out = ((topleft_out[-1] + topleft_out[1]) * 5 +
  200|  45.3k|                            topleft_out[0] * 6 + 8) >> 4;
  201|  1.91M|    }
  202|       |
  203|  13.6M|    return mode;
  204|  13.6M|}
dav1d_prepare_intra_edges_16bpc:
   86|  16.4M|{
   87|  16.4M|    const int bitdepth = bitdepth_from_max(bitdepth_max);
  ------------------
  |  |   75|  16.4M|#define bitdepth_from_max(bitdepth_max) (32 - clz(bitdepth_max))
  ------------------
   88|  16.4M|    assert(y < h && x < w);
  ------------------
  |  Branch (88:5): [True: 16.4M, False: 443]
  |  Branch (88:5): [True: 16.4M, False: 18.4E]
  ------------------
   89|       |
   90|  16.4M|    switch (mode) {
   91|   434k|    case VERT_PRED:
  ------------------
  |  Branch (91:5): [True: 434k, False: 15.9M]
  ------------------
   92|  1.37M|    case HOR_PRED:
  ------------------
  |  Branch (92:5): [True: 938k, False: 15.4M]
  ------------------
   93|  1.55M|    case DIAG_DOWN_LEFT_PRED:
  ------------------
  |  Branch (93:5): [True: 184k, False: 16.2M]
  ------------------
   94|  1.78M|    case DIAG_DOWN_RIGHT_PRED:
  ------------------
  |  Branch (94:5): [True: 225k, False: 16.1M]
  ------------------
   95|  1.86M|    case VERT_RIGHT_PRED:
  ------------------
  |  Branch (95:5): [True: 86.5k, False: 16.3M]
  ------------------
   96|  2.18M|    case HOR_DOWN_PRED:
  ------------------
  |  Branch (96:5): [True: 318k, False: 16.0M]
  ------------------
   97|  3.17M|    case HOR_UP_PRED:
  ------------------
  |  Branch (97:5): [True: 992k, False: 15.4M]
  ------------------
   98|  3.36M|    case VERT_LEFT_PRED: {
  ------------------
  |  Branch (98:5): [True: 190k, False: 16.2M]
  ------------------
   99|  3.36M|        *angle = av1_mode_to_angle_map[mode - VERT_PRED] + 3 * *angle;
  100|       |
  101|  3.36M|        if (*angle <= 90)
  ------------------
  |  Branch (101:13): [True: 744k, False: 2.62M]
  ------------------
  102|   744k|            mode = *angle < 90 && have_top ? Z1_PRED : VERT_PRED;
  ------------------
  |  Branch (102:20): [True: 493k, False: 250k]
  |  Branch (102:35): [True: 477k, False: 16.1k]
  ------------------
  103|  2.62M|        else if (*angle < 180)
  ------------------
  |  Branch (103:18): [True: 913k, False: 1.71M]
  ------------------
  104|   913k|            mode = Z2_PRED;
  105|  1.71M|        else
  106|  1.71M|            mode = *angle > 180 && have_left ? Z3_PRED : HOR_PRED;
  ------------------
  |  Branch (106:20): [True: 1.18M, False: 528k]
  |  Branch (106:36): [True: 1.17M, False: 8.89k]
  ------------------
  107|  3.36M|        break;
  108|  3.17M|    }
  109|  10.1M|    case DC_PRED:
  ------------------
  |  Branch (109:5): [True: 10.1M, False: 6.30M]
  ------------------
  110|  10.8M|    case PAETH_PRED:
  ------------------
  |  Branch (110:5): [True: 760k, False: 15.6M]
  ------------------
  111|  10.8M|        mode = av1_mode_conv[mode][have_left][have_top];
  112|  10.8M|        break;
  113|  2.29M|    default:
  ------------------
  |  Branch (113:5): [True: 2.29M, False: 14.1M]
  ------------------
  114|  2.29M|        break;
  115|  16.4M|    }
  116|       |
  117|  16.4M|    const pixel *dst_top;
  118|  16.4M|    if (have_top &&
  ------------------
  |  Branch (118:9): [True: 15.5M, False: 889k]
  ------------------
  119|  15.5M|        (av1_intra_prediction_edges[mode].needs_top ||
  ------------------
  |  Branch (119:10): [True: 13.8M, False: 1.65M]
  ------------------
  120|  1.65M|         av1_intra_prediction_edges[mode].needs_topleft ||
  ------------------
  |  Branch (120:10): [True: 1.13M, False: 519k]
  ------------------
  121|   519k|         (av1_intra_prediction_edges[mode].needs_left && !have_left)))
  ------------------
  |  Branch (121:11): [True: 519k, False: 5]
  |  Branch (121:58): [True: 9.36k, False: 509k]
  ------------------
  122|  15.0M|    {
  123|  15.0M|        if (prefilter_toplevel_sb_edge) {
  ------------------
  |  Branch (123:13): [True: 780k, False: 14.2M]
  ------------------
  124|   780k|            dst_top = &prefilter_toplevel_sb_edge[x * 4];
  125|  14.2M|        } else {
  126|  14.2M|            dst_top = &dst[-PXSTRIDE(stride)];
  127|  14.2M|        }
  128|  15.0M|    }
  129|       |
  130|  16.4M|    if (av1_intra_prediction_edges[mode].needs_left) {
  ------------------
  |  Branch (130:9): [True: 14.3M, False: 2.04M]
  ------------------
  131|  14.3M|        const int sz = th << 2;
  132|  14.3M|        pixel *const left = &topleft_out[-sz];
  133|       |
  134|  14.3M|        if (have_left) {
  ------------------
  |  Branch (134:13): [True: 14.3M, False: 35.9k]
  ------------------
  135|  14.3M|            const int px_have = imin(sz, (h - y) << 2);
  136|       |
  137|   106M|            for (int i = 0; i < px_have; i++)
  ------------------
  |  Branch (137:29): [True: 92.3M, False: 14.3M]
  ------------------
  138|  92.3M|                left[sz - 1 - i] = dst[PXSTRIDE(stride) * i - 1];
  139|  14.3M|            if (px_have < sz)
  ------------------
  |  Branch (139:17): [True: 36.0k, False: 14.2M]
  ------------------
  140|  36.0k|                pixel_set(left, left[sz - px_have], sz - px_have);
  141|  14.3M|        } else {
  142|  35.9k|            pixel_set(left, have_top ? *dst_top : ((1 << bitdepth) >> 1) + 1, sz);
  ------------------
  |  Branch (142:29): [True: 34.5k, False: 1.42k]
  ------------------
  143|  35.9k|        }
  144|       |
  145|  14.3M|        if (av1_intra_prediction_edges[mode].needs_bottomleft) {
  ------------------
  |  Branch (145:13): [True: 1.17M, False: 13.1M]
  ------------------
  146|  1.17M|            const int have_bottomleft = (!have_left || y + th >= h) ? 0 :
  ------------------
  |  Branch (146:42): [True: 45, False: 1.17M]
  |  Branch (146:56): [True: 35.5k, False: 1.13M]
  ------------------
  147|  1.17M|                                        (edge_flags & EDGE_I444_LEFT_HAS_BOTTOM);
  148|       |
  149|  1.17M|            if (have_bottomleft) {
  ------------------
  |  Branch (149:17): [True: 104k, False: 1.06M]
  ------------------
  150|   104k|                const int px_have = imin(sz, (h - y - th) << 2);
  151|       |
  152|   737k|                for (int i = 0; i < px_have; i++)
  ------------------
  |  Branch (152:33): [True: 632k, False: 104k]
  ------------------
  153|   632k|                    left[-(i + 1)] = dst[(sz + i) * PXSTRIDE(stride) - 1];
  154|   104k|                if (px_have < sz)
  ------------------
  |  Branch (154:21): [True: 536, False: 104k]
  ------------------
  155|    536|                    pixel_set(left - sz, left[-px_have], sz - px_have);
  156|  1.06M|            } else {
  157|  1.06M|                pixel_set(left - sz, left[0], sz);
  158|  1.06M|            }
  159|  1.17M|        }
  160|  14.3M|    }
  161|       |
  162|  16.4M|    if (av1_intra_prediction_edges[mode].needs_top) {
  ------------------
  |  Branch (162:9): [True: 14.0M, False: 2.39M]
  ------------------
  163|  14.0M|        const int sz = tw << 2;
  164|  14.0M|        pixel *const top = &topleft_out[1];
  165|       |
  166|  14.0M|        if (have_top) {
  ------------------
  |  Branch (166:13): [True: 13.8M, False: 127k]
  ------------------
  167|  13.8M|            const int px_have = imin(sz, (w - x) << 2);
  168|  13.8M|            pixel_copy(top, dst_top, px_have);
  ------------------
  |  |   65|  13.8M|#define pixel_copy(a, b, c) memcpy(a, b, (c) << 1)
  ------------------
  169|  13.8M|            if (px_have < sz)
  ------------------
  |  Branch (169:17): [True: 817k, False: 13.0M]
  ------------------
  170|   817k|                pixel_set(top + px_have, top[px_have - 1], sz - px_have);
  171|  13.8M|        } else {
  172|   127k|            pixel_set(top, have_left ? dst[-1] : ((1 << bitdepth) >> 1) - 1, sz);
  ------------------
  |  Branch (172:28): [True: 121k, False: 6.34k]
  ------------------
  173|   127k|        }
  174|       |
  175|  14.0M|        if (av1_intra_prediction_edges[mode].needs_topright) {
  ------------------
  |  Branch (175:13): [True: 476k, False: 13.5M]
  ------------------
  176|   476k|            const int have_topright = (!have_top || x + tw >= w) ? 0 :
  ------------------
  |  Branch (176:40): [True: 8, False: 476k]
  |  Branch (176:53): [True: 4.64k, False: 471k]
  ------------------
  177|   476k|                                      (edge_flags & EDGE_I444_TOP_HAS_RIGHT);
  178|       |
  179|   476k|            if (have_topright) {
  ------------------
  |  Branch (179:17): [True: 427k, False: 48.8k]
  ------------------
  180|   427k|                const int px_have = imin(sz, (w - x - tw) << 2);
  181|       |
  182|   427k|                pixel_copy(top + sz, &dst_top[sz], px_have);
  ------------------
  |  |   65|   427k|#define pixel_copy(a, b, c) memcpy(a, b, (c) << 1)
  ------------------
  183|   427k|                if (px_have < sz)
  ------------------
  |  Branch (183:21): [True: 1.55k, False: 425k]
  ------------------
  184|  1.55k|                    pixel_set(top + sz + px_have, top[sz + px_have - 1],
  185|  1.55k|                              sz - px_have);
  186|   427k|            } else {
  187|  48.8k|                pixel_set(top + sz, top[sz - 1], sz);
  188|  48.8k|            }
  189|   476k|        }
  190|  14.0M|    }
  191|       |
  192|  16.4M|    if (av1_intra_prediction_edges[mode].needs_topleft) {
  ------------------
  |  Branch (192:9): [True: 3.44M, False: 12.9M]
  ------------------
  193|  3.44M|        if (have_left)
  ------------------
  |  Branch (193:13): [True: 3.42M, False: 16.0k]
  ------------------
  194|  3.42M|            *topleft_out = have_top ? dst_top[-1] : dst[-1];
  ------------------
  |  Branch (194:28): [True: 3.36M, False: 65.2k]
  ------------------
  195|  16.0k|        else
  196|  16.0k|            *topleft_out = have_top ? *dst_top : (1 << bitdepth) >> 1;
  ------------------
  |  Branch (196:28): [True: 13.4k, False: 2.69k]
  ------------------
  197|       |
  198|  3.44M|        if (mode == Z2_PRED && tw + th >= 6 && filter_edge)
  ------------------
  |  Branch (198:13): [True: 908k, False: 2.53M]
  |  Branch (198:32): [True: 52.0k, False: 856k]
  |  Branch (198:48): [True: 48.2k, False: 3.77k]
  ------------------
  199|  48.2k|            *topleft_out = ((topleft_out[-1] + topleft_out[1]) * 5 +
  200|  48.2k|                            topleft_out[0] * 6 + 8) >> 4;
  201|  3.44M|    }
  202|       |
  203|  16.4M|    return mode;
  204|  16.4M|}

dav1d_intra_pred_dsp_init_8bpc:
  744|  3.46k|COLD void bitfn(dav1d_intra_pred_dsp_init)(Dav1dIntraPredDSPContext *const c) {
  745|  3.46k|    c->intra_pred[DC_PRED      ] = ipred_dc_c;
  746|  3.46k|    c->intra_pred[DC_128_PRED  ] = ipred_dc_128_c;
  747|  3.46k|    c->intra_pred[TOP_DC_PRED  ] = ipred_dc_top_c;
  748|  3.46k|    c->intra_pred[LEFT_DC_PRED ] = ipred_dc_left_c;
  749|  3.46k|    c->intra_pred[HOR_PRED     ] = ipred_h_c;
  750|  3.46k|    c->intra_pred[VERT_PRED    ] = ipred_v_c;
  751|  3.46k|    c->intra_pred[PAETH_PRED   ] = ipred_paeth_c;
  752|  3.46k|    c->intra_pred[SMOOTH_PRED  ] = ipred_smooth_c;
  753|  3.46k|    c->intra_pred[SMOOTH_V_PRED] = ipred_smooth_v_c;
  754|  3.46k|    c->intra_pred[SMOOTH_H_PRED] = ipred_smooth_h_c;
  755|  3.46k|    c->intra_pred[Z1_PRED      ] = ipred_z1_c;
  756|  3.46k|    c->intra_pred[Z2_PRED      ] = ipred_z2_c;
  757|  3.46k|    c->intra_pred[Z3_PRED      ] = ipred_z3_c;
  758|  3.46k|    c->intra_pred[FILTER_PRED  ] = ipred_filter_c;
  759|       |
  760|  3.46k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I420 - 1] = cfl_ac_420_c;
  761|  3.46k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I422 - 1] = cfl_ac_422_c;
  762|  3.46k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I444 - 1] = cfl_ac_444_c;
  763|       |
  764|  3.46k|    c->cfl_pred[DC_PRED     ] = ipred_cfl_c;
  765|  3.46k|    c->cfl_pred[DC_128_PRED ] = ipred_cfl_128_c;
  766|  3.46k|    c->cfl_pred[TOP_DC_PRED ] = ipred_cfl_top_c;
  767|  3.46k|    c->cfl_pred[LEFT_DC_PRED] = ipred_cfl_left_c;
  768|       |
  769|  3.46k|    c->pal_pred = pal_pred_c;
  770|       |
  771|  3.46k|#if HAVE_ASM
  772|       |#if ARCH_AARCH64 || ARCH_ARM
  773|       |    intra_pred_dsp_init_arm(c);
  774|       |#elif ARCH_RISCV
  775|       |    intra_pred_dsp_init_riscv(c);
  776|       |#elif ARCH_X86
  777|       |    intra_pred_dsp_init_x86(c);
  778|       |#elif ARCH_LOONGARCH64
  779|       |    intra_pred_dsp_init_loongarch(c);
  780|       |#endif
  781|  3.46k|#endif
  782|  3.46k|}
dav1d_intra_pred_dsp_init_16bpc:
  744|  5.11k|COLD void bitfn(dav1d_intra_pred_dsp_init)(Dav1dIntraPredDSPContext *const c) {
  745|  5.11k|    c->intra_pred[DC_PRED      ] = ipred_dc_c;
  746|  5.11k|    c->intra_pred[DC_128_PRED  ] = ipred_dc_128_c;
  747|  5.11k|    c->intra_pred[TOP_DC_PRED  ] = ipred_dc_top_c;
  748|  5.11k|    c->intra_pred[LEFT_DC_PRED ] = ipred_dc_left_c;
  749|  5.11k|    c->intra_pred[HOR_PRED     ] = ipred_h_c;
  750|  5.11k|    c->intra_pred[VERT_PRED    ] = ipred_v_c;
  751|  5.11k|    c->intra_pred[PAETH_PRED   ] = ipred_paeth_c;
  752|  5.11k|    c->intra_pred[SMOOTH_PRED  ] = ipred_smooth_c;
  753|  5.11k|    c->intra_pred[SMOOTH_V_PRED] = ipred_smooth_v_c;
  754|  5.11k|    c->intra_pred[SMOOTH_H_PRED] = ipred_smooth_h_c;
  755|  5.11k|    c->intra_pred[Z1_PRED      ] = ipred_z1_c;
  756|  5.11k|    c->intra_pred[Z2_PRED      ] = ipred_z2_c;
  757|  5.11k|    c->intra_pred[Z3_PRED      ] = ipred_z3_c;
  758|  5.11k|    c->intra_pred[FILTER_PRED  ] = ipred_filter_c;
  759|       |
  760|  5.11k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I420 - 1] = cfl_ac_420_c;
  761|  5.11k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I422 - 1] = cfl_ac_422_c;
  762|  5.11k|    c->cfl_ac[DAV1D_PIXEL_LAYOUT_I444 - 1] = cfl_ac_444_c;
  763|       |
  764|  5.11k|    c->cfl_pred[DC_PRED     ] = ipred_cfl_c;
  765|  5.11k|    c->cfl_pred[DC_128_PRED ] = ipred_cfl_128_c;
  766|  5.11k|    c->cfl_pred[TOP_DC_PRED ] = ipred_cfl_top_c;
  767|  5.11k|    c->cfl_pred[LEFT_DC_PRED] = ipred_cfl_left_c;
  768|       |
  769|  5.11k|    c->pal_pred = pal_pred_c;
  770|       |
  771|  5.11k|#if HAVE_ASM
  772|       |#if ARCH_AARCH64 || ARCH_ARM
  773|       |    intra_pred_dsp_init_arm(c);
  774|       |#elif ARCH_RISCV
  775|       |    intra_pred_dsp_init_riscv(c);
  776|       |#elif ARCH_X86
  777|       |    intra_pred_dsp_init_x86(c);
  778|       |#elif ARCH_LOONGARCH64
  779|       |    intra_pred_dsp_init_loongarch(c);
  780|       |#endif
  781|  5.11k|#endif
  782|  5.11k|}

itx_1d.c:inv_dct4_1d_internal_c:
   68|  1.67M|{
   69|  1.67M|    assert(stride > 0);
  ------------------
  |  Branch (69:5): [True: 1.67M, False: 2]
  ------------------
   70|  1.67M|    const int in0 = c[0 * stride], in1 = c[1 * stride];
   71|       |
   72|  1.67M|    int t0, t1, t2, t3;
   73|  1.67M|    if (tx64) {
  ------------------
  |  Branch (73:9): [True: 903k, False: 775k]
  ------------------
   74|   903k|        t0 = t1 = (in0 * 181 + 128) >> 8;
   75|   903k|        t2 = (in1 * 1567 + 2048) >> 12;
   76|   903k|        t3 = (in1 * 3784 + 2048) >> 12;
   77|   903k|    } else {
   78|   775k|        const int in2 = c[2 * stride], in3 = c[3 * stride];
   79|       |
   80|   775k|        t0 = ((in0 + in2) * 181 + 128) >> 8;
   81|   775k|        t1 = ((in0 - in2) * 181 + 128) >> 8;
   82|   775k|        t2 = ((in1 *  1567         - in3 * (3784 - 4096) + 2048) >> 12) - in3;
   83|   775k|        t3 = ((in1 * (3784 - 4096) + in3 *  1567         + 2048) >> 12) + in1;
   84|   775k|    }
   85|       |
   86|  1.67M|    c[0 * stride] = CLIP(t0 + t3);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
   87|  1.67M|    c[1 * stride] = CLIP(t1 + t2);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
   88|  1.67M|    c[2 * stride] = CLIP(t1 - t2);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
   89|  1.67M|    c[3 * stride] = CLIP(t0 - t3);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
   90|  1.67M|}
itx_1d.c:inv_dct8_1d_internal_c:
  101|  1.67M|{
  102|  1.67M|    assert(stride > 0);
  ------------------
  |  Branch (102:5): [True: 1.67M, False: 18.4E]
  ------------------
  103|  1.67M|    inv_dct4_1d_internal_c(c, stride << 1, min, max, tx64);
  104|       |
  105|  1.67M|    const int in1 = c[1 * stride], in3 = c[3 * stride];
  106|       |
  107|  1.67M|    int t4a, t5a, t6a, t7a;
  108|  1.67M|    if (tx64) {
  ------------------
  |  Branch (108:9): [True: 903k, False: 776k]
  ------------------
  109|   903k|        t4a = (in1 *   799 + 2048) >> 12;
  110|   903k|        t5a = (in3 * -2276 + 2048) >> 12;
  111|   903k|        t6a = (in3 *  3406 + 2048) >> 12;
  112|   903k|        t7a = (in1 *  4017 + 2048) >> 12;
  113|   903k|    } else {
  114|   776k|        const int in5 = c[5 * stride], in7 = c[7 * stride];
  115|       |
  116|   776k|        t4a = ((in1 *   799         - in7 * (4017 - 4096) + 2048) >> 12) - in7;
  117|   776k|        t5a =  (in5 *  1703         - in3 *  1138         + 1024) >> 11;
  118|   776k|        t6a =  (in5 *  1138         + in3 *  1703         + 1024) >> 11;
  119|   776k|        t7a = ((in1 * (4017 - 4096) + in7 *  799          + 2048) >> 12) + in1;
  120|   776k|    }
  121|       |
  122|  1.67M|    const int t4  = CLIP(t4a + t5a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  123|  1.67M|              t5a = CLIP(t4a - t5a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  124|  1.67M|    const int t7  = CLIP(t7a + t6a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  125|  1.67M|              t6a = CLIP(t7a - t6a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  126|       |
  127|  1.67M|    const int t5  = ((t6a - t5a) * 181 + 128) >> 8;
  128|  1.67M|    const int t6  = ((t6a + t5a) * 181 + 128) >> 8;
  129|       |
  130|  1.67M|    const int t0 = c[0 * stride];
  131|  1.67M|    const int t1 = c[2 * stride];
  132|  1.67M|    const int t2 = c[4 * stride];
  133|  1.67M|    const int t3 = c[6 * stride];
  134|       |
  135|  1.67M|    c[0 * stride] = CLIP(t0 + t7);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  136|  1.67M|    c[1 * stride] = CLIP(t1 + t6);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  137|  1.67M|    c[2 * stride] = CLIP(t2 + t5);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  138|  1.67M|    c[3 * stride] = CLIP(t3 + t4);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  139|  1.67M|    c[4 * stride] = CLIP(t3 - t4);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  140|  1.67M|    c[5 * stride] = CLIP(t2 - t5);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  141|  1.67M|    c[6 * stride] = CLIP(t1 - t6);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  142|  1.67M|    c[7 * stride] = CLIP(t0 - t7);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  143|  1.67M|}
itx_1d.c:inv_dct16_1d_c:
  242|   200k|{
  243|   200k|    inv_dct16_1d_internal_c(c, stride, min, max, 0);
  244|   200k|}
itx_1d.c:inv_dct16_1d_internal_c:
  154|  1.67M|{
  155|  1.67M|    assert(stride > 0);
  ------------------
  |  Branch (155:5): [True: 1.67M, False: 18.4E]
  ------------------
  156|  1.67M|    inv_dct8_1d_internal_c(c, stride << 1, min, max, tx64);
  157|       |
  158|  1.67M|    const int in1 = c[1 * stride], in3 = c[3 * stride];
  159|  1.67M|    const int in5 = c[5 * stride], in7 = c[7 * stride];
  160|       |
  161|  1.67M|    int t8a, t9a, t10a, t11a, t12a, t13a, t14a, t15a;
  162|  1.67M|    if (tx64) {
  ------------------
  |  Branch (162:9): [True: 904k, False: 775k]
  ------------------
  163|   904k|        t8a  = (in1 *   401 + 2048) >> 12;
  164|   904k|        t9a  = (in7 * -2598 + 2048) >> 12;
  165|   904k|        t10a = (in5 *  1931 + 2048) >> 12;
  166|   904k|        t11a = (in3 * -1189 + 2048) >> 12;
  167|   904k|        t12a = (in3 *  3920 + 2048) >> 12;
  168|   904k|        t13a = (in5 *  3612 + 2048) >> 12;
  169|   904k|        t14a = (in7 *  3166 + 2048) >> 12;
  170|   904k|        t15a = (in1 *  4076 + 2048) >> 12;
  171|   904k|    } else {
  172|   775k|        const int in9  = c[ 9 * stride], in11 = c[11 * stride];
  173|   775k|        const int in13 = c[13 * stride], in15 = c[15 * stride];
  174|       |
  175|   775k|        t8a  = ((in1  *   401         - in15 * (4076 - 4096) + 2048) >> 12) - in15;
  176|   775k|        t9a  =  (in9  *  1583         - in7  *  1299         + 1024) >> 11;
  177|   775k|        t10a = ((in5  *  1931         - in11 * (3612 - 4096) + 2048) >> 12) - in11;
  178|   775k|        t11a = ((in13 * (3920 - 4096) - in3  *  1189         + 2048) >> 12) + in13;
  179|   775k|        t12a = ((in13 *  1189         + in3  * (3920 - 4096) + 2048) >> 12) + in3;
  180|   775k|        t13a = ((in5  * (3612 - 4096) + in11 *  1931         + 2048) >> 12) + in5;
  181|   775k|        t14a =  (in9  *  1299         + in7  *  1583         + 1024) >> 11;
  182|   775k|        t15a = ((in1  * (4076 - 4096) + in15 *   401         + 2048) >> 12) + in1;
  183|   775k|    }
  184|       |
  185|  1.67M|    int t8  = CLIP(t8a  + t9a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  186|  1.67M|    int t9  = CLIP(t8a  - t9a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  187|  1.67M|    int t10 = CLIP(t11a - t10a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  188|  1.67M|    int t11 = CLIP(t11a + t10a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  189|  1.67M|    int t12 = CLIP(t12a + t13a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  190|  1.67M|    int t13 = CLIP(t12a - t13a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  191|  1.67M|    int t14 = CLIP(t15a - t14a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  192|  1.67M|    int t15 = CLIP(t15a + t14a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  193|       |
  194|  1.67M|    t9a  = ((  t14 *  1567         - t9  * (3784 - 4096)  + 2048) >> 12) - t9;
  195|  1.67M|    t14a = ((  t14 * (3784 - 4096) + t9  *  1567          + 2048) >> 12) + t14;
  196|  1.67M|    t10a = ((-(t13 * (3784 - 4096) + t10 *  1567)         + 2048) >> 12) - t13;
  197|  1.67M|    t13a = ((  t13 *  1567         - t10 * (3784 - 4096)  + 2048) >> 12) - t10;
  198|       |
  199|  1.67M|    t8a  = CLIP(t8   + t11);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  200|  1.67M|    t9   = CLIP(t9a  + t10a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  201|  1.67M|    t10  = CLIP(t9a  - t10a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  202|  1.67M|    t11a = CLIP(t8   - t11);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  203|  1.67M|    t12a = CLIP(t15  - t12);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  204|  1.67M|    t13  = CLIP(t14a - t13a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  205|  1.67M|    t14  = CLIP(t14a + t13a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  206|  1.67M|    t15a = CLIP(t15  + t12);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  207|       |
  208|  1.67M|    t10a = ((t13  - t10)  * 181 + 128) >> 8;
  209|  1.67M|    t13a = ((t13  + t10)  * 181 + 128) >> 8;
  210|  1.67M|    t11  = ((t12a - t11a) * 181 + 128) >> 8;
  211|  1.67M|    t12  = ((t12a + t11a) * 181 + 128) >> 8;
  212|       |
  213|  1.67M|    const int t0 = c[ 0 * stride];
  214|  1.67M|    const int t1 = c[ 2 * stride];
  215|  1.67M|    const int t2 = c[ 4 * stride];
  216|  1.67M|    const int t3 = c[ 6 * stride];
  217|  1.67M|    const int t4 = c[ 8 * stride];
  218|  1.67M|    const int t5 = c[10 * stride];
  219|  1.67M|    const int t6 = c[12 * stride];
  220|  1.67M|    const int t7 = c[14 * stride];
  221|       |
  222|  1.67M|    c[ 0 * stride] = CLIP(t0 + t15a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  223|  1.67M|    c[ 1 * stride] = CLIP(t1 + t14);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  224|  1.67M|    c[ 2 * stride] = CLIP(t2 + t13a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  225|  1.67M|    c[ 3 * stride] = CLIP(t3 + t12);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  226|  1.67M|    c[ 4 * stride] = CLIP(t4 + t11);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  227|  1.67M|    c[ 5 * stride] = CLIP(t5 + t10a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  228|  1.67M|    c[ 6 * stride] = CLIP(t6 + t9);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  229|  1.67M|    c[ 7 * stride] = CLIP(t7 + t8a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  230|  1.67M|    c[ 8 * stride] = CLIP(t7 - t8a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  231|  1.67M|    c[ 9 * stride] = CLIP(t6 - t9);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  232|  1.67M|    c[10 * stride] = CLIP(t5 - t10a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  233|  1.67M|    c[11 * stride] = CLIP(t4 - t11);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  234|  1.67M|    c[12 * stride] = CLIP(t3 - t12);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  235|  1.67M|    c[13 * stride] = CLIP(t2 - t13a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  236|  1.67M|    c[14 * stride] = CLIP(t1 - t14);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  237|  1.67M|    c[15 * stride] = CLIP(t0 - t15a);
  ------------------
  |  |   37|  1.67M|#define CLIP(a) iclip(a, min, max)
  ------------------
  238|  1.67M|}
itx_1d.c:inv_dct32_1d_c:
  432|   575k|{
  433|   575k|    inv_dct32_1d_internal_c(c, stride, min, max, 0);
  434|   575k|}
itx_1d.c:inv_dct32_1d_internal_c:
  249|  1.47M|{
  250|  1.47M|    assert(stride > 0);
  ------------------
  |  Branch (250:5): [True: 1.47M, False: 18.4E]
  ------------------
  251|  1.47M|    inv_dct16_1d_internal_c(c, stride << 1, min, max, tx64);
  252|       |
  253|  1.47M|    const int in1  = c[ 1 * stride], in3  = c[ 3 * stride];
  254|  1.47M|    const int in5  = c[ 5 * stride], in7  = c[ 7 * stride];
  255|  1.47M|    const int in9  = c[ 9 * stride], in11 = c[11 * stride];
  256|  1.47M|    const int in13 = c[13 * stride], in15 = c[15 * stride];
  257|       |
  258|  1.47M|    int t16a, t17a, t18a, t19a, t20a, t21a, t22a, t23a;
  259|  1.47M|    int t24a, t25a, t26a, t27a, t28a, t29a, t30a, t31a;
  260|  1.47M|    if (tx64) {
  ------------------
  |  Branch (260:9): [True: 905k, False: 574k]
  ------------------
  261|   905k|        t16a = (in1  *   201 + 2048) >> 12;
  262|   905k|        t17a = (in15 * -2751 + 2048) >> 12;
  263|   905k|        t18a = (in9  *  1751 + 2048) >> 12;
  264|   905k|        t19a = (in7  * -1380 + 2048) >> 12;
  265|   905k|        t20a = (in5  *   995 + 2048) >> 12;
  266|   905k|        t21a = (in11 * -2106 + 2048) >> 12;
  267|   905k|        t22a = (in13 *  2440 + 2048) >> 12;
  268|   905k|        t23a = (in3  *  -601 + 2048) >> 12;
  269|   905k|        t24a = (in3  *  4052 + 2048) >> 12;
  270|   905k|        t25a = (in13 *  3290 + 2048) >> 12;
  271|   905k|        t26a = (in11 *  3513 + 2048) >> 12;
  272|   905k|        t27a = (in5  *  3973 + 2048) >> 12;
  273|   905k|        t28a = (in7  *  3857 + 2048) >> 12;
  274|   905k|        t29a = (in9  *  3703 + 2048) >> 12;
  275|   905k|        t30a = (in15 *  3035 + 2048) >> 12;
  276|   905k|        t31a = (in1  *  4091 + 2048) >> 12;
  277|   905k|    } else {
  278|   574k|        const int in17 = c[17 * stride], in19 = c[19 * stride];
  279|   574k|        const int in21 = c[21 * stride], in23 = c[23 * stride];
  280|   574k|        const int in25 = c[25 * stride], in27 = c[27 * stride];
  281|   574k|        const int in29 = c[29 * stride], in31 = c[31 * stride];
  282|       |
  283|   574k|        t16a = ((in1  *   201         - in31 * (4091 - 4096) + 2048) >> 12) - in31;
  284|   574k|        t17a = ((in17 * (3035 - 4096) - in15 *  2751         + 2048) >> 12) + in17;
  285|   574k|        t18a = ((in9  *  1751         - in23 * (3703 - 4096) + 2048) >> 12) - in23;
  286|   574k|        t19a = ((in25 * (3857 - 4096) - in7  *  1380         + 2048) >> 12) + in25;
  287|   574k|        t20a = ((in5  *   995         - in27 * (3973 - 4096) + 2048) >> 12) - in27;
  288|   574k|        t21a = ((in21 * (3513 - 4096) - in11 *  2106         + 2048) >> 12) + in21;
  289|   574k|        t22a =  (in13 *  1220         - in19 *  1645         + 1024) >> 11;
  290|   574k|        t23a = ((in29 * (4052 - 4096) - in3  *   601         + 2048) >> 12) + in29;
  291|   574k|        t24a = ((in29 *   601         + in3  * (4052 - 4096) + 2048) >> 12) + in3;
  292|   574k|        t25a =  (in13 *  1645         + in19 *  1220         + 1024) >> 11;
  293|   574k|        t26a = ((in21 *  2106         + in11 * (3513 - 4096) + 2048) >> 12) + in11;
  294|   574k|        t27a = ((in5  * (3973 - 4096) + in27 *   995         + 2048) >> 12) + in5;
  295|   574k|        t28a = ((in25 *  1380         + in7  * (3857 - 4096) + 2048) >> 12) + in7;
  296|   574k|        t29a = ((in9  * (3703 - 4096) + in23 *  1751         + 2048) >> 12) + in9;
  297|   574k|        t30a = ((in17 *  2751         + in15 * (3035 - 4096) + 2048) >> 12) + in15;
  298|   574k|        t31a = ((in1  * (4091 - 4096) + in31 *   201         + 2048) >> 12) + in1;
  299|   574k|    }
  300|       |
  301|  1.47M|    int t16 = CLIP(t16a + t17a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  302|  1.47M|    int t17 = CLIP(t16a - t17a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  303|  1.47M|    int t18 = CLIP(t19a - t18a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  304|  1.47M|    int t19 = CLIP(t19a + t18a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  305|  1.47M|    int t20 = CLIP(t20a + t21a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  306|  1.47M|    int t21 = CLIP(t20a - t21a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  307|  1.47M|    int t22 = CLIP(t23a - t22a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  308|  1.47M|    int t23 = CLIP(t23a + t22a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  309|  1.47M|    int t24 = CLIP(t24a + t25a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  310|  1.47M|    int t25 = CLIP(t24a - t25a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  311|  1.47M|    int t26 = CLIP(t27a - t26a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  312|  1.47M|    int t27 = CLIP(t27a + t26a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  313|  1.47M|    int t28 = CLIP(t28a + t29a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  314|  1.47M|    int t29 = CLIP(t28a - t29a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  315|  1.47M|    int t30 = CLIP(t31a - t30a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  316|  1.47M|    int t31 = CLIP(t31a + t30a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  317|       |
  318|  1.47M|    t17a = ((  t30 *   799         - t17 * (4017 - 4096)  + 2048) >> 12) - t17;
  319|  1.47M|    t30a = ((  t30 * (4017 - 4096) + t17 *   799          + 2048) >> 12) + t30;
  320|  1.47M|    t18a = ((-(t29 * (4017 - 4096) + t18 *   799)         + 2048) >> 12) - t29;
  321|  1.47M|    t29a = ((  t29 *   799         - t18 * (4017 - 4096)  + 2048) >> 12) - t18;
  322|  1.47M|    t21a =  (  t26 *  1703         - t21 *  1138          + 1024) >> 11;
  323|  1.47M|    t26a =  (  t26 *  1138         + t21 *  1703          + 1024) >> 11;
  324|  1.47M|    t22a =  (-(t25 *  1138         + t22 *  1703        ) + 1024) >> 11;
  325|  1.47M|    t25a =  (  t25 *  1703         - t22 *  1138          + 1024) >> 11;
  326|       |
  327|  1.47M|    t16a = CLIP(t16  + t19);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  328|  1.47M|    t17  = CLIP(t17a + t18a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  329|  1.47M|    t18  = CLIP(t17a - t18a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  330|  1.47M|    t19a = CLIP(t16  - t19);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  331|  1.47M|    t20a = CLIP(t23  - t20);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  332|  1.47M|    t21  = CLIP(t22a - t21a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  333|  1.47M|    t22  = CLIP(t22a + t21a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  334|  1.47M|    t23a = CLIP(t23  + t20);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  335|  1.47M|    t24a = CLIP(t24  + t27);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  336|  1.47M|    t25  = CLIP(t25a + t26a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  337|  1.47M|    t26  = CLIP(t25a - t26a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  338|  1.47M|    t27a = CLIP(t24  - t27);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  339|  1.47M|    t28a = CLIP(t31  - t28);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  340|  1.47M|    t29  = CLIP(t30a - t29a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  341|  1.47M|    t30  = CLIP(t30a + t29a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  342|  1.47M|    t31a = CLIP(t31  + t28);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  343|       |
  344|  1.47M|    t18a = ((  t29  *  1567         - t18  * (3784 - 4096)  + 2048) >> 12) - t18;
  345|  1.47M|    t29a = ((  t29  * (3784 - 4096) + t18  *  1567          + 2048) >> 12) + t29;
  346|  1.47M|    t19  = ((  t28a *  1567         - t19a * (3784 - 4096)  + 2048) >> 12) - t19a;
  347|  1.47M|    t28  = ((  t28a * (3784 - 4096) + t19a *  1567          + 2048) >> 12) + t28a;
  348|  1.47M|    t20  = ((-(t27a * (3784 - 4096) + t20a *  1567)         + 2048) >> 12) - t27a;
  349|  1.47M|    t27  = ((  t27a *  1567         - t20a * (3784 - 4096)  + 2048) >> 12) - t20a;
  350|  1.47M|    t21a = ((-(t26  * (3784 - 4096) + t21  *  1567)         + 2048) >> 12) - t26;
  351|  1.47M|    t26a = ((  t26  *  1567         - t21  * (3784 - 4096)  + 2048) >> 12) - t21;
  352|       |
  353|  1.47M|    t16  = CLIP(t16a + t23a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  354|  1.47M|    t17a = CLIP(t17  + t22);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  355|  1.47M|    t18  = CLIP(t18a + t21a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  356|  1.47M|    t19a = CLIP(t19  + t20);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  357|  1.47M|    t20a = CLIP(t19  - t20);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  358|  1.47M|    t21  = CLIP(t18a - t21a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  359|  1.47M|    t22a = CLIP(t17  - t22);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  360|  1.47M|    t23  = CLIP(t16a - t23a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  361|  1.47M|    t24  = CLIP(t31a - t24a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  362|  1.47M|    t25a = CLIP(t30  - t25);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  363|  1.47M|    t26  = CLIP(t29a - t26a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  364|  1.47M|    t27a = CLIP(t28  - t27);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  365|  1.47M|    t28a = CLIP(t28  + t27);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  366|  1.47M|    t29  = CLIP(t29a + t26a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  367|  1.47M|    t30a = CLIP(t30  + t25);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  368|  1.47M|    t31  = CLIP(t31a + t24a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  369|       |
  370|  1.47M|    t20  = ((t27a - t20a) * 181 + 128) >> 8;
  371|  1.47M|    t27  = ((t27a + t20a) * 181 + 128) >> 8;
  372|  1.47M|    t21a = ((t26  - t21 ) * 181 + 128) >> 8;
  373|  1.47M|    t26a = ((t26  + t21 ) * 181 + 128) >> 8;
  374|  1.47M|    t22  = ((t25a - t22a) * 181 + 128) >> 8;
  375|  1.47M|    t25  = ((t25a + t22a) * 181 + 128) >> 8;
  376|  1.47M|    t23a = ((t24  - t23 ) * 181 + 128) >> 8;
  377|  1.47M|    t24a = ((t24  + t23 ) * 181 + 128) >> 8;
  378|       |
  379|  1.47M|    const int t0  = c[ 0 * stride];
  380|  1.47M|    const int t1  = c[ 2 * stride];
  381|  1.47M|    const int t2  = c[ 4 * stride];
  382|  1.47M|    const int t3  = c[ 6 * stride];
  383|  1.47M|    const int t4  = c[ 8 * stride];
  384|  1.47M|    const int t5  = c[10 * stride];
  385|  1.47M|    const int t6  = c[12 * stride];
  386|  1.47M|    const int t7  = c[14 * stride];
  387|  1.47M|    const int t8  = c[16 * stride];
  388|  1.47M|    const int t9  = c[18 * stride];
  389|  1.47M|    const int t10 = c[20 * stride];
  390|  1.47M|    const int t11 = c[22 * stride];
  391|  1.47M|    const int t12 = c[24 * stride];
  392|  1.47M|    const int t13 = c[26 * stride];
  393|  1.47M|    const int t14 = c[28 * stride];
  394|  1.47M|    const int t15 = c[30 * stride];
  395|       |
  396|  1.47M|    c[ 0 * stride] = CLIP(t0  + t31);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  397|  1.47M|    c[ 1 * stride] = CLIP(t1  + t30a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  398|  1.47M|    c[ 2 * stride] = CLIP(t2  + t29);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  399|  1.47M|    c[ 3 * stride] = CLIP(t3  + t28a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  400|  1.47M|    c[ 4 * stride] = CLIP(t4  + t27);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  401|  1.47M|    c[ 5 * stride] = CLIP(t5  + t26a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  402|  1.47M|    c[ 6 * stride] = CLIP(t6  + t25);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  403|  1.47M|    c[ 7 * stride] = CLIP(t7  + t24a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  404|  1.47M|    c[ 8 * stride] = CLIP(t8  + t23a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  405|  1.47M|    c[ 9 * stride] = CLIP(t9  + t22);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  406|  1.47M|    c[10 * stride] = CLIP(t10 + t21a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  407|  1.47M|    c[11 * stride] = CLIP(t11 + t20);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  408|  1.47M|    c[12 * stride] = CLIP(t12 + t19a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  409|  1.47M|    c[13 * stride] = CLIP(t13 + t18);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  410|  1.47M|    c[14 * stride] = CLIP(t14 + t17a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  411|  1.47M|    c[15 * stride] = CLIP(t15 + t16);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  412|  1.47M|    c[16 * stride] = CLIP(t15 - t16);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  413|  1.47M|    c[17 * stride] = CLIP(t14 - t17a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  414|  1.47M|    c[18 * stride] = CLIP(t13 - t18);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  415|  1.47M|    c[19 * stride] = CLIP(t12 - t19a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  416|  1.47M|    c[20 * stride] = CLIP(t11 - t20);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  417|  1.47M|    c[21 * stride] = CLIP(t10 - t21a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  418|  1.47M|    c[22 * stride] = CLIP(t9  - t22);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  419|  1.47M|    c[23 * stride] = CLIP(t8  - t23a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  420|  1.47M|    c[24 * stride] = CLIP(t7  - t24a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  421|  1.47M|    c[25 * stride] = CLIP(t6  - t25);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  422|  1.47M|    c[26 * stride] = CLIP(t5  - t26a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  423|  1.47M|    c[27 * stride] = CLIP(t4  - t27);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  424|  1.47M|    c[28 * stride] = CLIP(t3  - t28a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  425|  1.47M|    c[29 * stride] = CLIP(t2  - t29);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  426|  1.47M|    c[30 * stride] = CLIP(t1  - t30a);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  427|  1.47M|    c[31 * stride] = CLIP(t0  - t31);
  ------------------
  |  |   37|  1.47M|#define CLIP(a) iclip(a, min, max)
  ------------------
  428|  1.47M|}
itx_1d.c:inv_dct64_1d_c:
  438|   904k|{
  439|   904k|    assert(stride > 0);
  ------------------
  |  Branch (439:5): [True: 904k, False: 18.4E]
  ------------------
  440|   904k|    inv_dct32_1d_internal_c(c, stride << 1, min, max, 1);
  441|       |
  442|   904k|    const int in1  = c[ 1 * stride], in3  = c[ 3 * stride];
  443|   904k|    const int in5  = c[ 5 * stride], in7  = c[ 7 * stride];
  444|   904k|    const int in9  = c[ 9 * stride], in11 = c[11 * stride];
  445|   904k|    const int in13 = c[13 * stride], in15 = c[15 * stride];
  446|   904k|    const int in17 = c[17 * stride], in19 = c[19 * stride];
  447|   904k|    const int in21 = c[21 * stride], in23 = c[23 * stride];
  448|   904k|    const int in25 = c[25 * stride], in27 = c[27 * stride];
  449|   904k|    const int in29 = c[29 * stride], in31 = c[31 * stride];
  450|       |
  451|   904k|    int t32a = (in1  *   101 + 2048) >> 12;
  452|   904k|    int t33a = (in31 * -2824 + 2048) >> 12;
  453|   904k|    int t34a = (in17 *  1660 + 2048) >> 12;
  454|   904k|    int t35a = (in15 * -1474 + 2048) >> 12;
  455|   904k|    int t36a = (in9  *   897 + 2048) >> 12;
  456|   904k|    int t37a = (in23 * -2191 + 2048) >> 12;
  457|   904k|    int t38a = (in25 *  2359 + 2048) >> 12;
  458|   904k|    int t39a = (in7  *  -700 + 2048) >> 12;
  459|   904k|    int t40a = (in5  *   501 + 2048) >> 12;
  460|   904k|    int t41a = (in27 * -2520 + 2048) >> 12;
  461|   904k|    int t42a = (in21 *  2019 + 2048) >> 12;
  462|   904k|    int t43a = (in11 * -1092 + 2048) >> 12;
  463|   904k|    int t44a = (in13 *  1285 + 2048) >> 12;
  464|   904k|    int t45a = (in19 * -1842 + 2048) >> 12;
  465|   904k|    int t46a = (in29 *  2675 + 2048) >> 12;
  466|   904k|    int t47a = (in3  *  -301 + 2048) >> 12;
  467|   904k|    int t48a = (in3  *  4085 + 2048) >> 12;
  468|   904k|    int t49a = (in29 *  3102 + 2048) >> 12;
  469|   904k|    int t50a = (in19 *  3659 + 2048) >> 12;
  470|   904k|    int t51a = (in13 *  3889 + 2048) >> 12;
  471|   904k|    int t52a = (in11 *  3948 + 2048) >> 12;
  472|   904k|    int t53a = (in21 *  3564 + 2048) >> 12;
  473|   904k|    int t54a = (in27 *  3229 + 2048) >> 12;
  474|   904k|    int t55a = (in5  *  4065 + 2048) >> 12;
  475|   904k|    int t56a = (in7  *  4036 + 2048) >> 12;
  476|   904k|    int t57a = (in25 *  3349 + 2048) >> 12;
  477|   904k|    int t58a = (in23 *  3461 + 2048) >> 12;
  478|   904k|    int t59a = (in9  *  3996 + 2048) >> 12;
  479|   904k|    int t60a = (in15 *  3822 + 2048) >> 12;
  480|   904k|    int t61a = (in17 *  3745 + 2048) >> 12;
  481|   904k|    int t62a = (in31 *  2967 + 2048) >> 12;
  482|   904k|    int t63a = (in1  *  4095 + 2048) >> 12;
  483|       |
  484|   904k|    int t32 = CLIP(t32a + t33a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  485|   904k|    int t33 = CLIP(t32a - t33a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  486|   904k|    int t34 = CLIP(t35a - t34a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  487|   904k|    int t35 = CLIP(t35a + t34a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  488|   904k|    int t36 = CLIP(t36a + t37a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  489|   904k|    int t37 = CLIP(t36a - t37a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  490|   904k|    int t38 = CLIP(t39a - t38a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  491|   904k|    int t39 = CLIP(t39a + t38a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  492|   904k|    int t40 = CLIP(t40a + t41a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  493|   904k|    int t41 = CLIP(t40a - t41a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  494|   904k|    int t42 = CLIP(t43a - t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  495|   904k|    int t43 = CLIP(t43a + t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  496|   904k|    int t44 = CLIP(t44a + t45a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  497|   904k|    int t45 = CLIP(t44a - t45a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  498|   904k|    int t46 = CLIP(t47a - t46a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  499|   904k|    int t47 = CLIP(t47a + t46a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  500|   904k|    int t48 = CLIP(t48a + t49a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  501|   904k|    int t49 = CLIP(t48a - t49a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  502|   904k|    int t50 = CLIP(t51a - t50a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  503|   904k|    int t51 = CLIP(t51a + t50a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  504|   904k|    int t52 = CLIP(t52a + t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  505|   904k|    int t53 = CLIP(t52a - t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  506|   904k|    int t54 = CLIP(t55a - t54a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  507|   904k|    int t55 = CLIP(t55a + t54a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  508|   904k|    int t56 = CLIP(t56a + t57a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  509|   904k|    int t57 = CLIP(t56a - t57a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  510|   904k|    int t58 = CLIP(t59a - t58a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  511|   904k|    int t59 = CLIP(t59a + t58a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  512|   904k|    int t60 = CLIP(t60a + t61a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  513|   904k|    int t61 = CLIP(t60a - t61a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  514|   904k|    int t62 = CLIP(t63a - t62a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  515|   904k|    int t63 = CLIP(t63a + t62a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  516|       |
  517|   904k|    t33a = ((t33 * (4096 - 4076) + t62 *   401         + 2048) >> 12) - t33;
  518|   904k|    t34a = ((t34 *  -401         + t61 * (4096 - 4076) + 2048) >> 12) - t61;
  519|   904k|    t37a =  (t37 * -1299         + t58 *  1583         + 1024) >> 11;
  520|   904k|    t38a =  (t38 * -1583         + t57 * -1299         + 1024) >> 11;
  521|   904k|    t41a = ((t41 * (4096 - 3612) + t54 *  1931         + 2048) >> 12) - t41;
  522|   904k|    t42a = ((t42 * -1931         + t53 * (4096 - 3612) + 2048) >> 12) - t53;
  523|   904k|    t45a = ((t45 * -1189         + t50 * (3920 - 4096) + 2048) >> 12) + t50;
  524|   904k|    t46a = ((t46 * (4096 - 3920) + t49 * -1189         + 2048) >> 12) - t46;
  525|   904k|    t49a = ((t46 * -1189         + t49 * (3920 - 4096) + 2048) >> 12) + t49;
  526|   904k|    t50a = ((t45 * (3920 - 4096) + t50 *  1189         + 2048) >> 12) + t45;
  527|   904k|    t53a = ((t42 * (4096 - 3612) + t53 *  1931         + 2048) >> 12) - t42;
  528|   904k|    t54a = ((t41 *  1931         + t54 * (3612 - 4096) + 2048) >> 12) + t54;
  529|   904k|    t57a =  (t38 * -1299         + t57 *  1583         + 1024) >> 11;
  530|   904k|    t58a =  (t37 *  1583         + t58 *  1299         + 1024) >> 11;
  531|   904k|    t61a = ((t34 * (4096 - 4076) + t61 *   401         + 2048) >> 12) - t34;
  532|   904k|    t62a = ((t33 *   401         + t62 * (4076 - 4096) + 2048) >> 12) + t62;
  533|       |
  534|   904k|    t32a = CLIP(t32  + t35);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  535|   904k|    t33  = CLIP(t33a + t34a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  536|   904k|    t34  = CLIP(t33a - t34a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  537|   904k|    t35a = CLIP(t32  - t35);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  538|   904k|    t36a = CLIP(t39  - t36);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  539|   904k|    t37  = CLIP(t38a - t37a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  540|   904k|    t38  = CLIP(t38a + t37a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  541|   904k|    t39a = CLIP(t39  + t36);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  542|   904k|    t40a = CLIP(t40  + t43);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  543|   904k|    t41  = CLIP(t41a + t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  544|   904k|    t42  = CLIP(t41a - t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  545|   904k|    t43a = CLIP(t40  - t43);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  546|   904k|    t44a = CLIP(t47  - t44);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  547|   904k|    t45  = CLIP(t46a - t45a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  548|   904k|    t46  = CLIP(t46a + t45a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  549|   904k|    t47a = CLIP(t47  + t44);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  550|   904k|    t48a = CLIP(t48  + t51);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  551|   904k|    t49  = CLIP(t49a + t50a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  552|   904k|    t50  = CLIP(t49a - t50a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  553|   904k|    t51a = CLIP(t48  - t51);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  554|   904k|    t52a = CLIP(t55  - t52);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  555|   904k|    t53  = CLIP(t54a - t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  556|   904k|    t54  = CLIP(t54a + t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  557|   904k|    t55a = CLIP(t55  + t52);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  558|   904k|    t56a = CLIP(t56  + t59);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  559|   904k|    t57  = CLIP(t57a + t58a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  560|   904k|    t58  = CLIP(t57a - t58a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  561|   904k|    t59a = CLIP(t56  - t59);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  562|   904k|    t60a = CLIP(t63  - t60);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  563|   904k|    t61  = CLIP(t62a - t61a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  564|   904k|    t62  = CLIP(t62a + t61a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  565|   904k|    t63a = CLIP(t63  + t60);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  566|       |
  567|   904k|    t34a = ((t34  * (4096 - 4017) + t61  *   799         + 2048) >> 12) - t34;
  568|   904k|    t35  = ((t35a * (4096 - 4017) + t60a *   799         + 2048) >> 12) - t35a;
  569|   904k|    t36  = ((t36a *  -799         + t59a * (4096 - 4017) + 2048) >> 12) - t59a;
  570|   904k|    t37a = ((t37  *  -799         + t58  * (4096 - 4017) + 2048) >> 12) - t58;
  571|   904k|    t42a =  (t42  * -1138         + t53  *  1703         + 1024) >> 11;
  572|   904k|    t43  =  (t43a * -1138         + t52a *  1703         + 1024) >> 11;
  573|   904k|    t44  =  (t44a * -1703         + t51a * -1138         + 1024) >> 11;
  574|   904k|    t45a =  (t45  * -1703         + t50  * -1138         + 1024) >> 11;
  575|   904k|    t50a =  (t45  * -1138         + t50  *  1703         + 1024) >> 11;
  576|   904k|    t51  =  (t44a * -1138         + t51a *  1703         + 1024) >> 11;
  577|   904k|    t52  =  (t43a *  1703         + t52a *  1138         + 1024) >> 11;
  578|   904k|    t53a =  (t42  *  1703         + t53  *  1138         + 1024) >> 11;
  579|   904k|    t58a = ((t37  * (4096 - 4017) + t58  *   799         + 2048) >> 12) - t37;
  580|   904k|    t59  = ((t36a * (4096 - 4017) + t59a *   799         + 2048) >> 12) - t36a;
  581|   904k|    t60  = ((t35a *   799         + t60a * (4017 - 4096) + 2048) >> 12) + t60a;
  582|   904k|    t61a = ((t34  *   799         + t61  * (4017 - 4096) + 2048) >> 12) + t61;
  583|       |
  584|   904k|    t32  = CLIP(t32a + t39a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  585|   904k|    t33a = CLIP(t33  + t38);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  586|   904k|    t34  = CLIP(t34a + t37a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  587|   904k|    t35a = CLIP(t35  + t36);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  588|   904k|    t36a = CLIP(t35  - t36);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  589|   904k|    t37  = CLIP(t34a - t37a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  590|   904k|    t38a = CLIP(t33  - t38);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  591|   904k|    t39  = CLIP(t32a - t39a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  592|   904k|    t40  = CLIP(t47a - t40a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  593|   904k|    t41a = CLIP(t46  - t41);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  594|   904k|    t42  = CLIP(t45a - t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  595|   904k|    t43a = CLIP(t44  - t43);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  596|   904k|    t44a = CLIP(t44  + t43);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  597|   904k|    t45  = CLIP(t45a + t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  598|   904k|    t46a = CLIP(t46  + t41);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  599|   904k|    t47  = CLIP(t47a + t40a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  600|   904k|    t48  = CLIP(t48a + t55a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  601|   904k|    t49a = CLIP(t49  + t54);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  602|   904k|    t50  = CLIP(t50a + t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  603|   904k|    t51a = CLIP(t51  + t52);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  604|   904k|    t52a = CLIP(t51  - t52);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  605|   904k|    t53  = CLIP(t50a - t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  606|   904k|    t54a = CLIP(t49  - t54);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  607|   904k|    t55  = CLIP(t48a - t55a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  608|   904k|    t56  = CLIP(t63a - t56a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  609|   904k|    t57a = CLIP(t62  - t57);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  610|   904k|    t58  = CLIP(t61a - t58a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  611|   904k|    t59a = CLIP(t60  - t59);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  612|   904k|    t60a = CLIP(t60  + t59);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  613|   904k|    t61  = CLIP(t61a + t58a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  614|   904k|    t62a = CLIP(t62  + t57);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  615|   904k|    t63  = CLIP(t63a + t56a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  616|       |
  617|   904k|    t36  = ((t36a * (4096 - 3784) + t59a *  1567         + 2048) >> 12) - t36a;
  618|   904k|    t37a = ((t37  * (4096 - 3784) + t58  *  1567         + 2048) >> 12) - t37;
  619|   904k|    t38  = ((t38a * (4096 - 3784) + t57a *  1567         + 2048) >> 12) - t38a;
  620|   904k|    t39a = ((t39  * (4096 - 3784) + t56  *  1567         + 2048) >> 12) - t39;
  621|   904k|    t40a = ((t40  * -1567         + t55  * (4096 - 3784) + 2048) >> 12) - t55;
  622|   904k|    t41  = ((t41a * -1567         + t54a * (4096 - 3784) + 2048) >> 12) - t54a;
  623|   904k|    t42a = ((t42  * -1567         + t53  * (4096 - 3784) + 2048) >> 12) - t53;
  624|   904k|    t43  = ((t43a * -1567         + t52a * (4096 - 3784) + 2048) >> 12) - t52a;
  625|   904k|    t52  = ((t43a * (4096 - 3784) + t52a *  1567         + 2048) >> 12) - t43a;
  626|   904k|    t53a = ((t42  * (4096 - 3784) + t53  *  1567         + 2048) >> 12) - t42;
  627|   904k|    t54  = ((t41a * (4096 - 3784) + t54a *  1567         + 2048) >> 12) - t41a;
  628|   904k|    t55a = ((t40  * (4096 - 3784) + t55  *  1567         + 2048) >> 12) - t40;
  629|   904k|    t56a = ((t39  *  1567         + t56  * (3784 - 4096) + 2048) >> 12) + t56;
  630|   904k|    t57  = ((t38a *  1567         + t57a * (3784 - 4096) + 2048) >> 12) + t57a;
  631|   904k|    t58a = ((t37  *  1567         + t58  * (3784 - 4096) + 2048) >> 12) + t58;
  632|   904k|    t59  = ((t36a *  1567         + t59a * (3784 - 4096) + 2048) >> 12) + t59a;
  633|       |
  634|   904k|    t32a = CLIP(t32  + t47);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  635|   904k|    t33  = CLIP(t33a + t46a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  636|   904k|    t34a = CLIP(t34  + t45);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  637|   904k|    t35  = CLIP(t35a + t44a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  638|   904k|    t36a = CLIP(t36  + t43);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  639|   904k|    t37  = CLIP(t37a + t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  640|   904k|    t38a = CLIP(t38  + t41);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  641|   904k|    t39  = CLIP(t39a + t40a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  642|   904k|    t40  = CLIP(t39a - t40a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  643|   904k|    t41a = CLIP(t38  - t41);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  644|   904k|    t42  = CLIP(t37a - t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  645|   904k|    t43a = CLIP(t36  - t43);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  646|   904k|    t44  = CLIP(t35a - t44a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  647|   904k|    t45a = CLIP(t34  - t45);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  648|   904k|    t46  = CLIP(t33a - t46a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  649|   904k|    t47a = CLIP(t32  - t47);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  650|   904k|    t48a = CLIP(t63  - t48);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  651|   904k|    t49  = CLIP(t62a - t49a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  652|   904k|    t50a = CLIP(t61  - t50);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  653|   904k|    t51  = CLIP(t60a - t51a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  654|   904k|    t52a = CLIP(t59  - t52);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  655|   904k|    t53  = CLIP(t58a - t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  656|   904k|    t54a = CLIP(t57  - t54);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  657|   904k|    t55  = CLIP(t56a - t55a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  658|   904k|    t56  = CLIP(t56a + t55a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  659|   904k|    t57a = CLIP(t57  + t54);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  660|   904k|    t58  = CLIP(t58a + t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  661|   904k|    t59a = CLIP(t59  + t52);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  662|   904k|    t60  = CLIP(t60a + t51a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  663|   904k|    t61a = CLIP(t61  + t50);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  664|   904k|    t62  = CLIP(t62a + t49a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  665|   904k|    t63a = CLIP(t63  + t48);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  666|       |
  667|   904k|    t40a = ((t55  - t40 ) * 181 + 128) >> 8;
  668|   904k|    t41  = ((t54a - t41a) * 181 + 128) >> 8;
  669|   904k|    t42a = ((t53  - t42 ) * 181 + 128) >> 8;
  670|   904k|    t43  = ((t52a - t43a) * 181 + 128) >> 8;
  671|   904k|    t44a = ((t51  - t44 ) * 181 + 128) >> 8;
  672|   904k|    t45  = ((t50a - t45a) * 181 + 128) >> 8;
  673|   904k|    t46a = ((t49  - t46 ) * 181 + 128) >> 8;
  674|   904k|    t47  = ((t48a - t47a) * 181 + 128) >> 8;
  675|   904k|    t48  = ((t47a + t48a) * 181 + 128) >> 8;
  676|   904k|    t49a = ((t46  + t49 ) * 181 + 128) >> 8;
  677|   904k|    t50  = ((t45a + t50a) * 181 + 128) >> 8;
  678|   904k|    t51a = ((t44  + t51 ) * 181 + 128) >> 8;
  679|   904k|    t52  = ((t43a + t52a) * 181 + 128) >> 8;
  680|   904k|    t53a = ((t42  + t53 ) * 181 + 128) >> 8;
  681|   904k|    t54  = ((t41a + t54a) * 181 + 128) >> 8;
  682|   904k|    t55a = ((t40  + t55 ) * 181 + 128) >> 8;
  683|       |
  684|   904k|    const int t0  = c[ 0 * stride];
  685|   904k|    const int t1  = c[ 2 * stride];
  686|   904k|    const int t2  = c[ 4 * stride];
  687|   904k|    const int t3  = c[ 6 * stride];
  688|   904k|    const int t4  = c[ 8 * stride];
  689|   904k|    const int t5  = c[10 * stride];
  690|   904k|    const int t6  = c[12 * stride];
  691|   904k|    const int t7  = c[14 * stride];
  692|   904k|    const int t8  = c[16 * stride];
  693|   904k|    const int t9  = c[18 * stride];
  694|   904k|    const int t10 = c[20 * stride];
  695|   904k|    const int t11 = c[22 * stride];
  696|   904k|    const int t12 = c[24 * stride];
  697|   904k|    const int t13 = c[26 * stride];
  698|   904k|    const int t14 = c[28 * stride];
  699|   904k|    const int t15 = c[30 * stride];
  700|   904k|    const int t16 = c[32 * stride];
  701|   904k|    const int t17 = c[34 * stride];
  702|   904k|    const int t18 = c[36 * stride];
  703|   904k|    const int t19 = c[38 * stride];
  704|   904k|    const int t20 = c[40 * stride];
  705|   904k|    const int t21 = c[42 * stride];
  706|   904k|    const int t22 = c[44 * stride];
  707|   904k|    const int t23 = c[46 * stride];
  708|   904k|    const int t24 = c[48 * stride];
  709|   904k|    const int t25 = c[50 * stride];
  710|   904k|    const int t26 = c[52 * stride];
  711|   904k|    const int t27 = c[54 * stride];
  712|   904k|    const int t28 = c[56 * stride];
  713|   904k|    const int t29 = c[58 * stride];
  714|   904k|    const int t30 = c[60 * stride];
  715|   904k|    const int t31 = c[62 * stride];
  716|       |
  717|   904k|    c[ 0 * stride] = CLIP(t0  + t63a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  718|   904k|    c[ 1 * stride] = CLIP(t1  + t62);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  719|   904k|    c[ 2 * stride] = CLIP(t2  + t61a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  720|   904k|    c[ 3 * stride] = CLIP(t3  + t60);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  721|   904k|    c[ 4 * stride] = CLIP(t4  + t59a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  722|   904k|    c[ 5 * stride] = CLIP(t5  + t58);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  723|   904k|    c[ 6 * stride] = CLIP(t6  + t57a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  724|   904k|    c[ 7 * stride] = CLIP(t7  + t56);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  725|   904k|    c[ 8 * stride] = CLIP(t8  + t55a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  726|   904k|    c[ 9 * stride] = CLIP(t9  + t54);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  727|   904k|    c[10 * stride] = CLIP(t10 + t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  728|   904k|    c[11 * stride] = CLIP(t11 + t52);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  729|   904k|    c[12 * stride] = CLIP(t12 + t51a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  730|   904k|    c[13 * stride] = CLIP(t13 + t50);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  731|   904k|    c[14 * stride] = CLIP(t14 + t49a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  732|   904k|    c[15 * stride] = CLIP(t15 + t48);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  733|   904k|    c[16 * stride] = CLIP(t16 + t47);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  734|   904k|    c[17 * stride] = CLIP(t17 + t46a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  735|   904k|    c[18 * stride] = CLIP(t18 + t45);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  736|   904k|    c[19 * stride] = CLIP(t19 + t44a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  737|   904k|    c[20 * stride] = CLIP(t20 + t43);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  738|   904k|    c[21 * stride] = CLIP(t21 + t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  739|   904k|    c[22 * stride] = CLIP(t22 + t41);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  740|   904k|    c[23 * stride] = CLIP(t23 + t40a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  741|   904k|    c[24 * stride] = CLIP(t24 + t39);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  742|   904k|    c[25 * stride] = CLIP(t25 + t38a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  743|   904k|    c[26 * stride] = CLIP(t26 + t37);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  744|   904k|    c[27 * stride] = CLIP(t27 + t36a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  745|   904k|    c[28 * stride] = CLIP(t28 + t35);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  746|   904k|    c[29 * stride] = CLIP(t29 + t34a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  747|   904k|    c[30 * stride] = CLIP(t30 + t33);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  748|   904k|    c[31 * stride] = CLIP(t31 + t32a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  749|   904k|    c[32 * stride] = CLIP(t31 - t32a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  750|   904k|    c[33 * stride] = CLIP(t30 - t33);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  751|   904k|    c[34 * stride] = CLIP(t29 - t34a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  752|   904k|    c[35 * stride] = CLIP(t28 - t35);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  753|   904k|    c[36 * stride] = CLIP(t27 - t36a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  754|   904k|    c[37 * stride] = CLIP(t26 - t37);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  755|   904k|    c[38 * stride] = CLIP(t25 - t38a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  756|   904k|    c[39 * stride] = CLIP(t24 - t39);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  757|   904k|    c[40 * stride] = CLIP(t23 - t40a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  758|   904k|    c[41 * stride] = CLIP(t22 - t41);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  759|   904k|    c[42 * stride] = CLIP(t21 - t42a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  760|   904k|    c[43 * stride] = CLIP(t20 - t43);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  761|   904k|    c[44 * stride] = CLIP(t19 - t44a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  762|   904k|    c[45 * stride] = CLIP(t18 - t45);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  763|   904k|    c[46 * stride] = CLIP(t17 - t46a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  764|   904k|    c[47 * stride] = CLIP(t16 - t47);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  765|   904k|    c[48 * stride] = CLIP(t15 - t48);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  766|   904k|    c[49 * stride] = CLIP(t14 - t49a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  767|   904k|    c[50 * stride] = CLIP(t13 - t50);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  768|   904k|    c[51 * stride] = CLIP(t12 - t51a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  769|   904k|    c[52 * stride] = CLIP(t11 - t52);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  770|   904k|    c[53 * stride] = CLIP(t10 - t53a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  771|   904k|    c[54 * stride] = CLIP(t9  - t54);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  772|   904k|    c[55 * stride] = CLIP(t8  - t55a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  773|   904k|    c[56 * stride] = CLIP(t7  - t56);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  774|   904k|    c[57 * stride] = CLIP(t6  - t57a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  775|   904k|    c[58 * stride] = CLIP(t5  - t58);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  776|   904k|    c[59 * stride] = CLIP(t4  - t59a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  777|   904k|    c[60 * stride] = CLIP(t3  - t60);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  778|   904k|    c[61 * stride] = CLIP(t2  - t61a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  779|   904k|    c[62 * stride] = CLIP(t1  - t62);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  780|   904k|    c[63 * stride] = CLIP(t0  - t63a);
  ------------------
  |  |   37|   904k|#define CLIP(a) iclip(a, min, max)
  ------------------
  781|   904k|}

dav1d_itx_dsp_init_8bpc:
  220|  3.46k|COLD void bitfn(dav1d_itx_dsp_init)(Dav1dInvTxfmDSPContext *const c, int bpc) {
  221|  3.46k|#define assign_itx_all_fn64(w, h, pfx) \
  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  224|       |
  225|  3.46k|#define assign_itx_all_fn32(w, h, pfx) \
  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  229|       |
  230|  3.46k|#define assign_itx_all_fn16(w, h, pfx) \
  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  252|       |
  253|  3.46k|#define assign_itx_all_fn84(w, h, pfx) \
  254|  3.46k|    assign_itx_all_fn16(w, h, pfx); \
  255|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  256|  3.46k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  257|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  258|  3.46k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  259|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  260|  3.46k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  261|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  262|  3.46k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  263|  3.46k|
  264|  3.46k|#if !(HAVE_ASM && TRIM_DSP_FUNCTIONS && ( \
  265|  3.46k|  ARCH_AARCH64 || \
  266|  3.46k|  (ARCH_ARM && (defined(__ARM_NEON) || defined(__APPLE__) || defined(_WIN32))) \
  267|  3.46k|))
  268|  3.46k|    c->itxfm_add[TX_4X4][WHT_WHT] = inv_txfm_add_wht_wht_4x4_c;
  269|  3.46k|#endif
  270|  3.46k|    assign_itx_all_fn84( 4,  4, );
  ------------------
  |  |  254|  3.46k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.46k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.46k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.46k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.46k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  271|  3.46k|    assign_itx_all_fn84( 4,  8, R);
  ------------------
  |  |  254|  3.46k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.46k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.46k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.46k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.46k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  272|  3.46k|    assign_itx_all_fn84( 4, 16, R);
  ------------------
  |  |  254|  3.46k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.46k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.46k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.46k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.46k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  273|  3.46k|    assign_itx_all_fn84( 8,  4, R);
  ------------------
  |  |  254|  3.46k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.46k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.46k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.46k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.46k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  274|  3.46k|    assign_itx_all_fn84( 8,  8, );
  ------------------
  |  |  254|  3.46k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.46k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.46k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.46k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.46k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  275|  3.46k|    assign_itx_all_fn84( 8, 16, R);
  ------------------
  |  |  254|  3.46k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.46k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.46k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.46k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.46k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  276|  3.46k|    assign_itx_all_fn32( 8, 32, R);
  ------------------
  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  277|  3.46k|    assign_itx_all_fn84(16,  4, R);
  ------------------
  |  |  254|  3.46k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.46k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.46k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.46k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.46k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  278|  3.46k|    assign_itx_all_fn84(16,  8, R);
  ------------------
  |  |  254|  3.46k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  3.46k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  3.46k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  3.46k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  3.46k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  279|  3.46k|    assign_itx_all_fn16(16, 16, );
  ------------------
  |  |  231|  3.46k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  ------------------
  |  |  232|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  233|  3.46k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  234|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  235|  3.46k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  236|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  237|  3.46k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  238|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  239|  3.46k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  240|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  241|  3.46k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  242|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  243|  3.46k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  244|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  245|  3.46k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  246|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  247|  3.46k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  248|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  249|  3.46k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  250|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  251|  3.46k|        inv_txfm_add_identity_dct_##w##x##h##_c
  ------------------
  280|  3.46k|    assign_itx_all_fn32(16, 32, R);
  ------------------
  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  281|  3.46k|    assign_itx_all_fn64(16, 64, R);
  ------------------
  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  282|  3.46k|    assign_itx_all_fn32(32,  8, R);
  ------------------
  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  283|  3.46k|    assign_itx_all_fn32(32, 16, R);
  ------------------
  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  284|  3.46k|    assign_itx_all_fn32(32, 32, );
  ------------------
  |  |  226|  3.46k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  3.46k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  285|  3.46k|    assign_itx_all_fn64(32, 64, R);
  ------------------
  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  286|  3.46k|    assign_itx_all_fn64(64, 16, R);
  ------------------
  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  287|  3.46k|    assign_itx_all_fn64(64, 32, R);
  ------------------
  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  288|  3.46k|    assign_itx_all_fn64(64, 64, );
  ------------------
  |  |  222|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  3.46k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  289|       |
  290|  3.46k|    int all_simd = 0;
  291|  3.46k|#if HAVE_ASM
  292|       |#if ARCH_AARCH64 || ARCH_ARM
  293|       |    itx_dsp_init_arm(c, bpc, &all_simd);
  294|       |#endif
  295|       |#if ARCH_LOONGARCH64
  296|       |    itx_dsp_init_loongarch(c, bpc);
  297|       |#endif
  298|       |#if ARCH_PPC64LE
  299|       |    itx_dsp_init_ppc(c, bpc);
  300|       |#endif
  301|       |#if ARCH_RISCV
  302|       |    itx_dsp_init_riscv(c, bpc);
  303|       |#endif
  304|  3.46k|#if ARCH_X86
  305|  3.46k|    itx_dsp_init_x86(c, bpc, &all_simd);
  306|  3.46k|#endif
  307|  3.46k|#endif
  308|       |
  309|  3.46k|    if (!all_simd)
  ------------------
  |  Branch (309:9): [True: 0, False: 3.46k]
  ------------------
  310|      0|        dav1d_init_last_nonzero_col_from_eob_tables();
  311|  3.46k|}
itx_tmpl.c:inv_txfm_add_c:
   47|  55.6k|{
   48|  55.6k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[tx];
   49|  55.6k|    const int w = 4 * t_dim->w, h = 4 * t_dim->h;
   50|  55.6k|    const int has_dconly = txtp == DCT_DCT;
   51|  55.6k|    assert(w >= 4 && w <= 64);
  ------------------
  |  Branch (51:5): [True: 55.6k, False: 18.4E]
  |  Branch (51:5): [True: 55.6k, False: 18.4E]
  ------------------
   52|  55.6k|    assert(h >= 4 && h <= 64);
  ------------------
  |  Branch (52:5): [True: 55.6k, False: 1]
  |  Branch (52:5): [True: 55.6k, False: 0]
  ------------------
   53|  55.6k|    assert(eob >= 0);
  ------------------
  |  Branch (53:5): [True: 55.6k, False: 0]
  ------------------
   54|       |
   55|  55.6k|    const int is_rect2 = w * 2 == h || h * 2 == w;
  ------------------
  |  Branch (55:26): [True: 7.40k, False: 48.2k]
  |  Branch (55:40): [True: 8.99k, False: 39.2k]
  ------------------
   56|  55.6k|    const int rnd = (1 << shift) >> 1;
   57|       |
   58|  55.6k|    if (eob < has_dconly) {
  ------------------
  |  Branch (58:9): [True: 24.7k, False: 30.9k]
  ------------------
   59|  24.7k|        int dc = coeff[0];
   60|  24.7k|        coeff[0] = 0;
   61|  24.7k|        if (is_rect2)
  ------------------
  |  Branch (61:13): [True: 6.99k, False: 17.7k]
  ------------------
   62|  6.99k|            dc = (dc * 181 + 128) >> 8;
   63|  24.7k|        dc = (dc * 181 + 128) >> 8;
   64|  24.7k|        dc = (dc + rnd) >> shift;
   65|  24.7k|        dc = (dc * 181 + 128 + 2048) >> 12;
   66|  1.03M|        for (int y = 0; y < h; y++, dst += PXSTRIDE(stride))
  ------------------
  |  |   53|  1.01M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (66:25): [True: 1.01M, False: 24.7k]
  ------------------
   67|  49.5M|            for (int x = 0; x < w; x++)
  ------------------
  |  Branch (67:29): [True: 48.5M, False: 1.01M]
  ------------------
   68|  48.5M|                dst[x] = iclip_pixel(dst[x] + dc);
  ------------------
  |  |   49|  48.5M|#define iclip_pixel iclip_u8
  ------------------
   69|  24.7k|        return;
   70|  24.7k|    }
   71|       |
   72|  30.9k|    const uint8_t *const txtps = dav1d_tx1d_types[txtp];
   73|  30.9k|    const itx_1d_fn first_1d_fn = dav1d_tx1d_fns[t_dim->lw][txtps[0]];
   74|  30.9k|    const itx_1d_fn second_1d_fn = dav1d_tx1d_fns[t_dim->lh][txtps[1]];
   75|  30.9k|    const int sh = imin(h, 32), sw = imin(w, 32);
   76|  30.9k|#if BITDEPTH == 8
   77|  30.9k|    const int row_clip_min = INT16_MIN;
   78|  30.9k|    const int col_clip_min = INT16_MIN;
   79|       |#else
   80|       |    const int row_clip_min = (int) ((unsigned) ~bitdepth_max << 7);
   81|       |    const int col_clip_min = (int) ((unsigned) ~bitdepth_max << 5);
   82|       |#endif
   83|  30.9k|    const int row_clip_max = ~row_clip_min;
   84|  30.9k|    const int col_clip_max = ~col_clip_min;
   85|       |
   86|  30.9k|    int32_t tmp[64 * 64], *c = tmp;
   87|  30.9k|    int last_nonzero_col; // in first 1d itx
   88|  30.9k|    if (txtps[1] == IDENTITY && txtps[0] != IDENTITY) {
  ------------------
  |  Branch (88:9): [True: 0, False: 30.9k]
  |  Branch (88:33): [True: 0, False: 0]
  ------------------
   89|      0|        last_nonzero_col = imin(sh - 1, eob);
   90|  30.9k|    } else if (txtps[0] == IDENTITY && txtps[1] != IDENTITY) {
  ------------------
  |  Branch (90:16): [True: 0, False: 30.9k]
  |  Branch (90:40): [True: 0, False: 0]
  ------------------
   91|      0|        last_nonzero_col = eob >> (t_dim->lw + 2);
   92|  30.9k|    } else {
   93|  30.9k|        last_nonzero_col = dav1d_last_nonzero_col_from_eob[tx][eob];
   94|  30.9k|    }
   95|  30.9k|    assert(last_nonzero_col < sh);
  ------------------
  |  Branch (95:5): [True: 30.9k, False: 18.4E]
  ------------------
   96|   401k|    for (int y = 0; y <= last_nonzero_col; y++, c += w) {
  ------------------
  |  Branch (96:21): [True: 370k, False: 30.9k]
  ------------------
   97|   370k|        if (is_rect2)
  ------------------
  |  Branch (97:13): [True: 101k, False: 269k]
  ------------------
   98|  2.90M|            for (int x = 0; x < sw; x++)
  ------------------
  |  Branch (98:29): [True: 2.80M, False: 101k]
  ------------------
   99|  2.80M|                c[x] = (coeff[y + x * sh] * 181 + 128) >> 8;
  100|   269k|        else
  101|  8.67M|            for (int x = 0; x < sw; x++)
  ------------------
  |  Branch (101:29): [True: 8.40M, False: 269k]
  ------------------
  102|  8.40M|                c[x] = coeff[y + x * sh];
  103|   370k|        first_1d_fn(c, 1, row_clip_min, row_clip_max);
  104|   370k|    }
  105|  30.9k|    if (last_nonzero_col + 1 < sh)
  ------------------
  |  Branch (105:9): [True: 25.9k, False: 5.04k]
  ------------------
  106|  25.9k|        memset(c, 0, sizeof(*c) * (sh - last_nonzero_col - 1) * w);
  107|       |
  108|  30.9k|    memset(coeff, 0, sizeof(*coeff) * sw * sh);
  109|  39.3M|    for (int i = 0; i < w * sh; i++)
  ------------------
  |  Branch (109:21): [True: 39.2M, False: 30.9k]
  ------------------
  110|  39.2M|        tmp[i] = iclip((tmp[i] + rnd) >> shift, col_clip_min, col_clip_max);
  111|       |
  112|  1.34M|    for (int x = 0; x < w; x++)
  ------------------
  |  Branch (112:21): [True: 1.30M, False: 30.9k]
  ------------------
  113|  1.30M|        second_1d_fn(&tmp[x], w, col_clip_min, col_clip_max);
  114|       |
  115|  30.9k|    c = tmp;
  116|  1.38M|    for (int y = 0; y < h; y++, dst += PXSTRIDE(stride))
  ------------------
  |  |   53|  1.35M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (116:21): [True: 1.35M, False: 30.9k]
  ------------------
  117|  63.5M|        for (int x = 0; x < w; x++)
  ------------------
  |  Branch (117:25): [True: 62.1M, False: 1.35M]
  ------------------
  118|  62.1M|            dst[x] = iclip_pixel(dst[x] + ((*c++ + 8) >> 4));
  ------------------
  |  |   49|  62.1M|#define iclip_pixel iclip_u8
  ------------------
  119|  30.9k|}
itx_tmpl.c:inv_txfm_add_dct_dct_16x32_c:
  127|  4.45k|                                               HIGHBD_DECL_SUFFIX) \
  128|  4.45k|{ \
  129|  4.45k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  4.45k|                   HIGHBD_TAIL_SUFFIX); \
  131|  4.45k|}
itx_tmpl.c:inv_txfm_add_dct_dct_16x64_c:
  127|  1.26k|                                               HIGHBD_DECL_SUFFIX) \
  128|  1.26k|{ \
  129|  1.26k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  1.26k|                   HIGHBD_TAIL_SUFFIX); \
  131|  1.26k|}
itx_tmpl.c:inv_txfm_add_dct_dct_32x16_c:
  127|  6.32k|                                               HIGHBD_DECL_SUFFIX) \
  128|  6.32k|{ \
  129|  6.32k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  6.32k|                   HIGHBD_TAIL_SUFFIX); \
  131|  6.32k|}
itx_tmpl.c:inv_txfm_add_dct_dct_32x32_c:
  127|  18.8k|                                               HIGHBD_DECL_SUFFIX) \
  128|  18.8k|{ \
  129|  18.8k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  18.8k|                   HIGHBD_TAIL_SUFFIX); \
  131|  18.8k|}
itx_tmpl.c:inv_txfm_add_dct_dct_32x64_c:
  127|  2.94k|                                               HIGHBD_DECL_SUFFIX) \
  128|  2.94k|{ \
  129|  2.94k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  2.94k|                   HIGHBD_TAIL_SUFFIX); \
  131|  2.94k|}
itx_tmpl.c:inv_txfm_add_dct_dct_64x16_c:
  127|  1.02k|                                               HIGHBD_DECL_SUFFIX) \
  128|  1.02k|{ \
  129|  1.02k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  1.02k|                   HIGHBD_TAIL_SUFFIX); \
  131|  1.02k|}
itx_tmpl.c:inv_txfm_add_dct_dct_64x32_c:
  127|  2.68k|                                               HIGHBD_DECL_SUFFIX) \
  128|  2.68k|{ \
  129|  2.68k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  2.68k|                   HIGHBD_TAIL_SUFFIX); \
  131|  2.68k|}
itx_tmpl.c:inv_txfm_add_dct_dct_64x64_c:
  127|  18.1k|                                               HIGHBD_DECL_SUFFIX) \
  128|  18.1k|{ \
  129|  18.1k|    inv_txfm_add_c(dst, stride, coeff, eob, pfx##TX_##w##X##h, shift, type \
  130|  18.1k|                   HIGHBD_TAIL_SUFFIX); \
  131|  18.1k|}
dav1d_itx_dsp_init_16bpc:
  220|  5.11k|COLD void bitfn(dav1d_itx_dsp_init)(Dav1dInvTxfmDSPContext *const c, int bpc) {
  221|  5.11k|#define assign_itx_all_fn64(w, h, pfx) \
  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  224|       |
  225|  5.11k|#define assign_itx_all_fn32(w, h, pfx) \
  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  229|       |
  230|  5.11k|#define assign_itx_all_fn16(w, h, pfx) \
  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  252|       |
  253|  5.11k|#define assign_itx_all_fn84(w, h, pfx) \
  254|  5.11k|    assign_itx_all_fn16(w, h, pfx); \
  255|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  256|  5.11k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  257|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  258|  5.11k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  259|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  260|  5.11k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  261|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  262|  5.11k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  263|  5.11k|
  264|  5.11k|#if !(HAVE_ASM && TRIM_DSP_FUNCTIONS && ( \
  265|  5.11k|  ARCH_AARCH64 || \
  266|  5.11k|  (ARCH_ARM && (defined(__ARM_NEON) || defined(__APPLE__) || defined(_WIN32))) \
  267|  5.11k|))
  268|  5.11k|    c->itxfm_add[TX_4X4][WHT_WHT] = inv_txfm_add_wht_wht_4x4_c;
  269|  5.11k|#endif
  270|  5.11k|    assign_itx_all_fn84( 4,  4, );
  ------------------
  |  |  254|  5.11k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  5.11k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  5.11k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  5.11k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  5.11k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  271|  5.11k|    assign_itx_all_fn84( 4,  8, R);
  ------------------
  |  |  254|  5.11k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  5.11k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  5.11k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  5.11k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  5.11k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  272|  5.11k|    assign_itx_all_fn84( 4, 16, R);
  ------------------
  |  |  254|  5.11k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  5.11k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  5.11k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  5.11k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  5.11k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  273|  5.11k|    assign_itx_all_fn84( 8,  4, R);
  ------------------
  |  |  254|  5.11k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  5.11k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  5.11k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  5.11k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  5.11k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  274|  5.11k|    assign_itx_all_fn84( 8,  8, );
  ------------------
  |  |  254|  5.11k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  5.11k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  5.11k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  5.11k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  5.11k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  275|  5.11k|    assign_itx_all_fn84( 8, 16, R);
  ------------------
  |  |  254|  5.11k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  5.11k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  5.11k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  5.11k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  5.11k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  276|  5.11k|    assign_itx_all_fn32( 8, 32, R);
  ------------------
  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  277|  5.11k|    assign_itx_all_fn84(16,  4, R);
  ------------------
  |  |  254|  5.11k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  5.11k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  5.11k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  5.11k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  5.11k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  278|  5.11k|    assign_itx_all_fn84(16,  8, R);
  ------------------
  |  |  254|  5.11k|    assign_itx_all_fn16(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  |  |  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  |  |  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  |  |  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  |  |  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  |  |  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  |  |  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  |  |  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  |  |  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  |  |  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  |  |  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  |  |  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  |  |  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  |  |  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  |  |  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  |  |  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  |  |  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  |  |  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  |  |  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  |  |  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  |  |  ------------------
  |  |  255|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_FLIPADST] = \
  |  |  256|  5.11k|        inv_txfm_add_flipadst_identity_##w##x##h##_c; \
  |  |  257|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_FLIPADST] = \
  |  |  258|  5.11k|        inv_txfm_add_identity_flipadst_##w##x##h##_c; \
  |  |  259|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_ADST] = \
  |  |  260|  5.11k|        inv_txfm_add_adst_identity_##w##x##h##_c; \
  |  |  261|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_ADST] = \
  |  |  262|  5.11k|        inv_txfm_add_identity_adst_##w##x##h##_c; \
  ------------------
  279|  5.11k|    assign_itx_all_fn16(16, 16, );
  ------------------
  |  |  231|  5.11k|    assign_itx_all_fn32(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  |  |  ------------------
  |  |  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  |  |  ------------------
  |  |  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  |  |  ------------------
  |  |  232|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_ADST ] = \
  |  |  233|  5.11k|        inv_txfm_add_adst_dct_##w##x##h##_c; \
  |  |  234|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_DCT ] = \
  |  |  235|  5.11k|        inv_txfm_add_dct_adst_##w##x##h##_c; \
  |  |  236|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_ADST] = \
  |  |  237|  5.11k|        inv_txfm_add_adst_adst_##w##x##h##_c; \
  |  |  238|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][ADST_FLIPADST] = \
  |  |  239|  5.11k|        inv_txfm_add_flipadst_adst_##w##x##h##_c; \
  |  |  240|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_ADST] = \
  |  |  241|  5.11k|        inv_txfm_add_adst_flipadst_##w##x##h##_c; \
  |  |  242|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_FLIPADST] = \
  |  |  243|  5.11k|        inv_txfm_add_flipadst_dct_##w##x##h##_c; \
  |  |  244|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_DCT] = \
  |  |  245|  5.11k|        inv_txfm_add_dct_flipadst_##w##x##h##_c; \
  |  |  246|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][FLIPADST_FLIPADST] = \
  |  |  247|  5.11k|        inv_txfm_add_flipadst_flipadst_##w##x##h##_c; \
  |  |  248|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][H_DCT] = \
  |  |  249|  5.11k|        inv_txfm_add_dct_identity_##w##x##h##_c; \
  |  |  250|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][V_DCT] = \
  |  |  251|  5.11k|        inv_txfm_add_identity_dct_##w##x##h##_c
  ------------------
  280|  5.11k|    assign_itx_all_fn32(16, 32, R);
  ------------------
  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  281|  5.11k|    assign_itx_all_fn64(16, 64, R);
  ------------------
  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  282|  5.11k|    assign_itx_all_fn32(32,  8, R);
  ------------------
  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  283|  5.11k|    assign_itx_all_fn32(32, 16, R);
  ------------------
  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  284|  5.11k|    assign_itx_all_fn32(32, 32, );
  ------------------
  |  |  226|  5.11k|    assign_itx_all_fn64(w, h, pfx); \
  |  |  ------------------
  |  |  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  |  |  ------------------
  |  |  227|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][IDTX] = \
  |  |  228|  5.11k|        inv_txfm_add_identity_identity_##w##x##h##_c
  ------------------
  285|  5.11k|    assign_itx_all_fn64(32, 64, R);
  ------------------
  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  286|  5.11k|    assign_itx_all_fn64(64, 16, R);
  ------------------
  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  287|  5.11k|    assign_itx_all_fn64(64, 32, R);
  ------------------
  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  288|  5.11k|    assign_itx_all_fn64(64, 64, );
  ------------------
  |  |  222|  5.11k|    c->itxfm_add[pfx##TX_##w##X##h][DCT_DCT  ] = \
  |  |  223|  5.11k|        inv_txfm_add_dct_dct_##w##x##h##_c
  ------------------
  289|       |
  290|  5.11k|    int all_simd = 0;
  291|  5.11k|#if HAVE_ASM
  292|       |#if ARCH_AARCH64 || ARCH_ARM
  293|       |    itx_dsp_init_arm(c, bpc, &all_simd);
  294|       |#endif
  295|       |#if ARCH_LOONGARCH64
  296|       |    itx_dsp_init_loongarch(c, bpc);
  297|       |#endif
  298|       |#if ARCH_PPC64LE
  299|       |    itx_dsp_init_ppc(c, bpc);
  300|       |#endif
  301|       |#if ARCH_RISCV
  302|       |    itx_dsp_init_riscv(c, bpc);
  303|       |#endif
  304|  5.11k|#if ARCH_X86
  305|  5.11k|    itx_dsp_init_x86(c, bpc, &all_simd);
  306|  5.11k|#endif
  307|  5.11k|#endif
  308|       |
  309|  5.11k|    if (!all_simd)
  ------------------
  |  Branch (309:9): [True: 2.99k, False: 2.11k]
  ------------------
  310|  2.99k|        dav1d_init_last_nonzero_col_from_eob_tables();
  311|  5.11k|}

dav1d_copy_lpf_8bpc:
  106|   241k|{
  107|   241k|    const int have_tt = f->c->n_tc > 1;
  108|   241k|    const int resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
  109|   241k|    const int offset = 8 * !!sby;
  110|   241k|    const ptrdiff_t *const src_stride = f->cur.stride;
  111|   241k|    const ptrdiff_t *const lr_stride = f->sr_cur.p.stride;
  112|   241k|    const int tt_off = have_tt * sby * (4 << f->seq_hdr->sb128);
  113|   241k|    pixel *const dst[3] = {
  114|   241k|        f->lf.lr_lpf_line[0] + tt_off * PXSTRIDE(lr_stride[0]),
  ------------------
  |  |   53|   241k|#define PXSTRIDE(x) (x)
  ------------------
  115|   241k|        f->lf.lr_lpf_line[1] + tt_off * PXSTRIDE(lr_stride[1]),
  ------------------
  |  |   53|   241k|#define PXSTRIDE(x) (x)
  ------------------
  116|   241k|        f->lf.lr_lpf_line[2] + tt_off * PXSTRIDE(lr_stride[1])
  ------------------
  |  |   53|   241k|#define PXSTRIDE(x) (x)
  ------------------
  117|   241k|    };
  118|       |
  119|       |    // TODO Also check block level restore type to reduce copying.
  120|   241k|    const int restore_planes = f->lf.restore_planes;
  121|       |
  122|   241k|    if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_Y) {
  ------------------
  |  Branch (122:9): [True: 72.3k, False: 168k]
  |  Branch (122:29): [True: 166k, False: 2.40k]
  ------------------
  123|   238k|        const int h = f->cur.p.h;
  124|   238k|        const int w = f->bw << 2;
  125|   238k|        const int row_h = imin((sby + 1) << (6 + f->seq_hdr->sb128), h - 1);
  126|   238k|        const int y_stripe = (sby << (6 + f->seq_hdr->sb128)) - offset;
  127|   238k|        if (restore_planes & LR_RESTORE_Y || !resize)
  ------------------
  |  Branch (127:13): [True: 171k, False: 66.8k]
  |  Branch (127:46): [True: 59.4k, False: 7.35k]
  ------------------
  128|   231k|            backup_lpf(f, dst[0], lr_stride[0],
  129|   231k|                       src[0] - offset * PXSTRIDE(src_stride[0]), src_stride[0],
  ------------------
  |  |   53|   231k|#define PXSTRIDE(x) (x)
  ------------------
  130|   231k|                       0, f->seq_hdr->sb128, y_stripe, row_h, w, h, 0, 1);
  131|   238k|        if (have_tt && resize) {
  ------------------
  |  Branch (131:13): [True: 238k, False: 158]
  |  Branch (131:24): [True: 9.98k, False: 228k]
  ------------------
  132|  9.98k|            const ptrdiff_t cdef_off_y = sby * 4 * PXSTRIDE(src_stride[0]);
  ------------------
  |  |   53|  9.98k|#define PXSTRIDE(x) (x)
  ------------------
  133|  9.98k|            backup_lpf(f, f->lf.cdef_lpf_line[0] + cdef_off_y, src_stride[0],
  134|  9.98k|                       src[0] - offset * PXSTRIDE(src_stride[0]), src_stride[0],
  ------------------
  |  |   53|  9.98k|#define PXSTRIDE(x) (x)
  ------------------
  135|  9.98k|                       0, f->seq_hdr->sb128, y_stripe, row_h, w, h, 0, 0);
  136|  9.98k|        }
  137|   238k|    }
  138|   241k|    if ((f->seq_hdr->cdef || restore_planes & (LR_RESTORE_U | LR_RESTORE_V)) &&
  ------------------
  |  Branch (138:10): [True: 72.3k, False: 168k]
  |  Branch (138:30): [True: 4.56k, False: 164k]
  ------------------
  139|  76.7k|        f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400)
  ------------------
  |  Branch (139:9): [True: 48.1k, False: 28.5k]
  ------------------
  140|  48.1k|    {
  141|  48.1k|        const int ss_ver = f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  142|  48.1k|        const int ss_hor = f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  143|  48.1k|        const int h = (f->cur.p.h + ss_ver) >> ss_ver;
  144|  48.1k|        const int w = f->bw << (2 - ss_hor);
  145|  48.1k|        const int row_h = imin((sby + 1) << ((6 - ss_ver) + f->seq_hdr->sb128), h - 1);
  146|  48.1k|        const int offset_uv = offset >> ss_ver;
  147|  48.1k|        const int y_stripe = (sby << ((6 - ss_ver) + f->seq_hdr->sb128)) - offset_uv;
  148|  48.1k|        const ptrdiff_t cdef_off_uv = sby * 4 * PXSTRIDE(src_stride[1]);
  ------------------
  |  |   53|  48.1k|#define PXSTRIDE(x) (x)
  ------------------
  149|  48.1k|        if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_U) {
  ------------------
  |  Branch (149:13): [True: 43.6k, False: 4.56k]
  |  Branch (149:33): [True: 3.39k, False: 1.16k]
  ------------------
  150|  47.0k|            if (restore_planes & LR_RESTORE_U || !resize)
  ------------------
  |  Branch (150:17): [True: 6.22k, False: 40.7k]
  |  Branch (150:50): [True: 34.1k, False: 6.64k]
  ------------------
  151|  40.3k|                backup_lpf(f, dst[1], lr_stride[1],
  152|  40.3k|                           src[1] - offset_uv * PXSTRIDE(src_stride[1]),
  ------------------
  |  |   53|  40.3k|#define PXSTRIDE(x) (x)
  ------------------
  153|  40.3k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  154|  40.3k|                           row_h, w, h, ss_hor, 1);
  155|  47.0k|            if (have_tt && resize)
  ------------------
  |  Branch (155:17): [True: 47.0k, False: 18.4E]
  |  Branch (155:28): [True: 10.7k, False: 36.2k]
  ------------------
  156|  10.7k|                backup_lpf(f, f->lf.cdef_lpf_line[1] + cdef_off_uv, src_stride[1],
  157|  10.7k|                           src[1] - offset_uv * PXSTRIDE(src_stride[1]),
  ------------------
  |  |   53|  10.7k|#define PXSTRIDE(x) (x)
  ------------------
  158|  10.7k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  159|  10.7k|                           row_h, w, h, ss_hor, 0);
  160|  47.0k|        }
  161|  48.1k|        if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_V) {
  ------------------
  |  Branch (161:13): [True: 43.6k, False: 4.56k]
  |  Branch (161:33): [True: 3.59k, False: 971]
  ------------------
  162|  47.2k|            if (restore_planes & LR_RESTORE_V || !resize)
  ------------------
  |  Branch (162:17): [True: 6.21k, False: 41.0k]
  |  Branch (162:50): [True: 33.8k, False: 7.15k]
  ------------------
  163|  40.0k|                backup_lpf(f, dst[2], lr_stride[1],
  164|  40.0k|                           src[2] - offset_uv * PXSTRIDE(src_stride[1]),
  ------------------
  |  |   53|  40.0k|#define PXSTRIDE(x) (x)
  ------------------
  165|  40.0k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  166|  40.0k|                           row_h, w, h, ss_hor, 1);
  167|  47.2k|            if (have_tt && resize)
  ------------------
  |  Branch (167:17): [True: 47.2k, False: 0]
  |  Branch (167:28): [True: 10.7k, False: 36.4k]
  ------------------
  168|  10.7k|                backup_lpf(f, f->lf.cdef_lpf_line[2] + cdef_off_uv, src_stride[1],
  169|  10.7k|                           src[2] - offset_uv * PXSTRIDE(src_stride[1]),
  ------------------
  |  |   53|  10.7k|#define PXSTRIDE(x) (x)
  ------------------
  170|  10.7k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  171|  10.7k|                           row_h, w, h, ss_hor, 0);
  172|  47.2k|        }
  173|  48.1k|    }
  174|   241k|}
dav1d_loopfilter_sbrow_cols_8bpc:
  316|   518k|{
  317|   518k|    int x, have_left;
  318|       |    // Don't filter outside the frame
  319|   518k|    const int is_sb64 = !f->seq_hdr->sb128;
  320|   518k|    const int starty4 = (sby & is_sb64) << 4;
  321|   518k|    const int sbsz = 32 >> is_sb64;
  322|   518k|    const int sbl2 = 5 - is_sb64;
  323|   518k|    const int halign = (f->bh + 31) & ~31;
  324|   518k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  325|   518k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  326|   518k|    const int vmask = 16 >> ss_ver, hmask = 16 >> ss_hor;
  327|   518k|    const unsigned vmax = 1U << vmask, hmax = 1U << hmask;
  328|   518k|    const unsigned endy4 = starty4 + imin(f->h4 - sby * sbsz, sbsz);
  329|   518k|    const unsigned uv_endy4 = (endy4 + ss_ver) >> ss_ver;
  330|       |
  331|       |    // fix lpf strength at tile col boundaries
  332|   518k|    const uint8_t *lpf_y = &f->lf.tx_lpf_right_edge[0][sby << sbl2];
  333|   518k|    const uint8_t *lpf_uv = &f->lf.tx_lpf_right_edge[1][sby << (sbl2 - ss_ver)];
  334|   822k|    for (int tile_col = 1;; tile_col++) {
  335|   822k|        x = f->frame_hdr->tiling.col_start_sb[tile_col];
  336|   822k|        if ((x << sbl2) >= f->bw) break;
  ------------------
  |  Branch (336:13): [True: 518k, False: 303k]
  ------------------
  337|   303k|        const int bx4 = x & is_sb64 ? 16 : 0, cbx4 = bx4 >> ss_hor;
  ------------------
  |  Branch (337:25): [True: 299k, False: 4.03k]
  ------------------
  338|   303k|        x >>= is_sb64;
  339|       |
  340|   303k|        uint16_t (*const y_hmask)[2] = lflvl[x].filter_y[0][bx4];
  341|  5.17M|        for (unsigned y = starty4, mask = 1 << y; y < endy4; y++, mask <<= 1) {
  ------------------
  |  Branch (341:51): [True: 4.87M, False: 303k]
  ------------------
  342|  4.87M|            const int sidx = mask >= 0x10000U;
  343|  4.87M|            const unsigned smask = mask >> (sidx << 4);
  344|  4.87M|            const int idx = 2 * !!(y_hmask[2][sidx] & smask) +
  345|  4.87M|                                !!(y_hmask[1][sidx] & smask);
  346|  4.87M|            y_hmask[2][sidx] &= ~smask;
  347|  4.87M|            y_hmask[1][sidx] &= ~smask;
  348|  4.87M|            y_hmask[0][sidx] &= ~smask;
  349|  4.87M|            y_hmask[imin(idx, lpf_y[y - starty4])][sidx] |= smask;
  350|  4.87M|        }
  351|       |
  352|   303k|        if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (352:13): [True: 5.43k, False: 298k]
  ------------------
  353|  5.43k|            uint16_t (*const uv_hmask)[2] = lflvl[x].filter_uv[0][cbx4];
  354|  62.4k|            for (unsigned y = starty4 >> ss_ver, uv_mask = 1 << y; y < uv_endy4;
  ------------------
  |  Branch (354:68): [True: 56.9k, False: 5.43k]
  ------------------
  355|  56.9k|                 y++, uv_mask <<= 1)
  356|  56.9k|            {
  357|  56.9k|                const int sidx = uv_mask >= vmax;
  358|  56.9k|                const unsigned smask = uv_mask >> (sidx << (4 - ss_ver));
  359|  56.9k|                const int idx = !!(uv_hmask[1][sidx] & smask);
  360|  56.9k|                uv_hmask[1][sidx] &= ~smask;
  361|  56.9k|                uv_hmask[0][sidx] &= ~smask;
  362|  56.9k|                uv_hmask[imin(idx, lpf_uv[y - (starty4 >> ss_ver)])][sidx] |= smask;
  363|  56.9k|            }
  364|  5.43k|        }
  365|   303k|        lpf_y  += halign;
  366|   303k|        lpf_uv += halign >> ss_ver;
  367|   303k|    }
  368|       |
  369|       |    // fix lpf strength at tile row boundaries
  370|   518k|    if (start_of_tile_row) {
  ------------------
  |  Branch (370:9): [True: 544, False: 518k]
  ------------------
  371|    544|        const BlockContext *a;
  372|    544|        for (x = 0, a = &f->a[f->sb128w * (start_of_tile_row - 1)];
  373|  2.55k|             x < f->sb128w; x++, a++)
  ------------------
  |  Branch (373:14): [True: 2.01k, False: 544]
  ------------------
  374|  2.01k|        {
  375|  2.01k|            uint16_t (*const y_vmask)[2] = lflvl[x].filter_y[1][starty4];
  376|  2.01k|            const unsigned w = imin(32, f->w4 - (x << 5));
  377|  61.0k|            for (unsigned mask = 1, i = 0; i < w; mask <<= 1, i++) {
  ------------------
  |  Branch (377:44): [True: 59.0k, False: 2.01k]
  ------------------
  378|  59.0k|                const int sidx = mask >= 0x10000U;
  379|  59.0k|                const unsigned smask = mask >> (sidx << 4);
  380|  59.0k|                const int idx = 2 * !!(y_vmask[2][sidx] & smask) +
  381|  59.0k|                                    !!(y_vmask[1][sidx] & smask);
  382|  59.0k|                y_vmask[2][sidx] &= ~smask;
  383|  59.0k|                y_vmask[1][sidx] &= ~smask;
  384|  59.0k|                y_vmask[0][sidx] &= ~smask;
  385|  59.0k|                y_vmask[imin(idx, a->tx_lpf_y[i])][sidx] |= smask;
  386|  59.0k|            }
  387|       |
  388|  2.01k|            if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (388:17): [True: 1.24k, False: 764]
  ------------------
  389|  1.24k|                const unsigned cw = (w + ss_hor) >> ss_hor;
  390|  1.24k|                uint16_t (*const uv_vmask)[2] = lflvl[x].filter_uv[1][starty4 >> ss_ver];
  391|  20.1k|                for (unsigned uv_mask = 1, i = 0; i < cw; uv_mask <<= 1, i++) {
  ------------------
  |  Branch (391:51): [True: 18.9k, False: 1.24k]
  ------------------
  392|  18.9k|                    const int sidx = uv_mask >= hmax;
  393|  18.9k|                    const unsigned smask = uv_mask >> (sidx << (4 - ss_hor));
  394|  18.9k|                    const int idx = !!(uv_vmask[1][sidx] & smask);
  395|  18.9k|                    uv_vmask[1][sidx] &= ~smask;
  396|  18.9k|                    uv_vmask[0][sidx] &= ~smask;
  397|  18.9k|                    uv_vmask[imin(idx, a->tx_lpf_uv[i])][sidx] |= smask;
  398|  18.9k|                }
  399|  1.24k|            }
  400|  2.01k|        }
  401|    544|    }
  402|       |
  403|   518k|    pixel *ptr;
  404|   518k|    uint8_t (*level_ptr)[4] = f->lf.level + f->b4_stride * sby * sbsz;
  405|  1.07M|    for (ptr = p[0], have_left = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (405:44): [True: 551k, False: 518k]
  ------------------
  406|   551k|         x++, have_left = 1, ptr += 128, level_ptr += 32)
  407|   551k|    {
  408|   551k|        filter_plane_cols_y(f, have_left, level_ptr, f->b4_stride,
  409|   551k|                            lflvl[x].filter_y[0], ptr, f->cur.stride[0],
  410|   551k|                            imin(32, f->w4 - x * 32), starty4, endy4);
  411|   551k|    }
  412|       |
  413|   518k|    if (!f->frame_hdr->loopfilter.level_u && !f->frame_hdr->loopfilter.level_v)
  ------------------
  |  Branch (413:9): [True: 489k, False: 29.2k]
  |  Branch (413:46): [True: 485k, False: 4.22k]
  ------------------
  414|   485k|        return;
  415|       |
  416|  33.4k|    ptrdiff_t uv_off;
  417|  33.4k|    level_ptr = f->lf.level + f->b4_stride * (sby * sbsz >> ss_ver);
  418|  92.5k|    for (uv_off = 0, have_left = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (418:44): [True: 59.1k, False: 33.4k]
  ------------------
  419|  59.1k|         x++, have_left = 1, uv_off += 128 >> ss_hor, level_ptr += 32 >> ss_hor)
  420|  59.1k|    {
  421|  59.1k|        filter_plane_cols_uv(f, have_left, level_ptr, f->b4_stride,
  422|  59.1k|                             lflvl[x].filter_uv[0],
  423|  59.1k|                             &p[1][uv_off], &p[2][uv_off], f->cur.stride[1],
  424|  59.1k|                             (imin(32, f->w4 - x * 32) + ss_hor) >> ss_hor,
  425|  59.1k|                             starty4 >> ss_ver, uv_endy4, ss_ver);
  426|  59.1k|    }
  427|  33.4k|}
dav1d_loopfilter_sbrow_rows_8bpc:
  432|   518k|{
  433|   518k|    int x;
  434|       |    // Don't filter outside the frame
  435|   518k|    const int have_top = sby > 0;
  436|   518k|    const int is_sb64 = !f->seq_hdr->sb128;
  437|   518k|    const int starty4 = (sby & is_sb64) << 4;
  438|   518k|    const int sbsz = 32 >> is_sb64;
  439|   518k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  440|   518k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  441|   518k|    const unsigned endy4 = starty4 + imin(f->h4 - sby * sbsz, sbsz);
  442|   518k|    const unsigned uv_endy4 = (endy4 + ss_ver) >> ss_ver;
  443|       |
  444|   518k|    pixel *ptr;
  445|   518k|    uint8_t (*level_ptr)[4] = f->lf.level + f->b4_stride * sby * sbsz;
  446|  1.07M|    for (ptr = p[0], x = 0; x < f->sb128w; x++, ptr += 128, level_ptr += 32) {
  ------------------
  |  Branch (446:29): [True: 551k, False: 518k]
  ------------------
  447|   551k|        filter_plane_rows_y(f, have_top, level_ptr, f->b4_stride,
  448|   551k|                            lflvl[x].filter_y[1], ptr, f->cur.stride[0],
  449|   551k|                            imin(32, f->w4 - x * 32), starty4, endy4);
  450|   551k|    }
  451|       |
  452|   518k|    if (!f->frame_hdr->loopfilter.level_u && !f->frame_hdr->loopfilter.level_v)
  ------------------
  |  Branch (452:9): [True: 489k, False: 29.2k]
  |  Branch (452:46): [True: 485k, False: 4.20k]
  ------------------
  453|   485k|        return;
  454|       |
  455|  33.4k|    ptrdiff_t uv_off;
  456|  33.4k|    level_ptr = f->lf.level + f->b4_stride * (sby * sbsz >> ss_ver);
  457|  92.5k|    for (uv_off = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (457:29): [True: 59.0k, False: 33.4k]
  ------------------
  458|  59.0k|         x++, uv_off += 128 >> ss_hor, level_ptr += 32 >> ss_hor)
  459|  59.0k|    {
  460|  59.0k|        filter_plane_rows_uv(f, have_top, level_ptr, f->b4_stride,
  461|  59.0k|                             lflvl[x].filter_uv[1],
  462|  59.0k|                             &p[1][uv_off], &p[2][uv_off], f->cur.stride[1],
  463|  59.0k|                             (imin(32, f->w4 - x * 32) + ss_hor) >> ss_hor,
  464|  59.0k|                             starty4 >> ss_ver, uv_endy4, ss_hor);
  465|  59.0k|    }
  466|  33.4k|}
lf_apply_tmpl.c:backup_lpf:
   47|  1.56M|{
   48|  1.56M|    const int cdef_backup = !lr_backup;
   49|  1.56M|    const int dst_w = f->frame_hdr->super_res.enabled ?
  ------------------
  |  Branch (49:23): [True: 109k, False: 1.45M]
  ------------------
   50|  1.45M|                      (f->frame_hdr->width[1] + ss_hor) >> ss_hor : src_w;
   51|       |
   52|       |    // The first stripe of the frame is shorter by 8 luma pixel rows.
   53|  1.56M|    int stripe_h = ((64 << (cdef_backup & sb128)) - 8 * !row) >> ss_ver;
   54|  1.56M|    src += (stripe_h - 2) * PXSTRIDE(src_stride);
  ------------------
  |  |   53|  1.56M|#define PXSTRIDE(x) (x)
  ------------------
   55|       |
   56|  1.56M|    if (f->c->n_tc == 1) {
  ------------------
  |  Branch (56:9): [True: 0, False: 1.56M]
  ------------------
   57|      0|        if (row) {
  ------------------
  |  Branch (57:13): [True: 0, False: 0]
  ------------------
   58|      0|            const int top = 4 << sb128;
   59|       |            // Copy the top part of the stored loop filtered pixels from the
   60|       |            // previous sb row needed above the first stripe of this sb row.
   61|      0|            pixel_copy(&dst[PXSTRIDE(dst_stride) *  0],
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
                          pixel_copy(&dst[PXSTRIDE(dst_stride) *  0],
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
   62|      0|                       &dst[PXSTRIDE(dst_stride) *  top],      dst_w);
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
   63|      0|            pixel_copy(&dst[PXSTRIDE(dst_stride) *  1],
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
                          pixel_copy(&dst[PXSTRIDE(dst_stride) *  1],
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
   64|      0|                       &dst[PXSTRIDE(dst_stride) * (top + 1)], dst_w);
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
   65|      0|            pixel_copy(&dst[PXSTRIDE(dst_stride) *  2],
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
                          pixel_copy(&dst[PXSTRIDE(dst_stride) *  2],
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
   66|      0|                       &dst[PXSTRIDE(dst_stride) * (top + 2)], dst_w);
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
   67|      0|            pixel_copy(&dst[PXSTRIDE(dst_stride) *  3],
  ------------------
  |  |   47|      0|#define pixel_copy memcpy
  ------------------
                          pixel_copy(&dst[PXSTRIDE(dst_stride) *  3],
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
   68|      0|                       &dst[PXSTRIDE(dst_stride) * (top + 3)], dst_w);
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
   69|      0|        }
   70|      0|        dst += 4 * PXSTRIDE(dst_stride);
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
   71|      0|    }
   72|       |
   73|  1.56M|    if (lr_backup && (f->frame_hdr->width[0] != f->frame_hdr->width[1])) {
  ------------------
  |  Branch (73:9): [True: 1.48M, False: 74.9k]
  |  Branch (73:22): [True: 29.7k, False: 1.45M]
  ------------------
   74|  68.2k|        while (row + stripe_h <= row_h) {
  ------------------
  |  Branch (74:16): [True: 38.4k, False: 29.7k]
  ------------------
   75|  38.4k|            const int n_lines = 4 - (row + stripe_h + 1 == h);
   76|  38.4k|            f->dsp->mc.resize(dst, dst_stride, src, src_stride,
   77|  38.4k|                              dst_w, n_lines, src_w, f->resize_step[ss_hor],
   78|  38.4k|                              f->resize_start[ss_hor] HIGHBD_CALL_SUFFIX);
   79|  38.4k|            row += stripe_h; // unmodified stripe_h for the 1st stripe
   80|  38.4k|            stripe_h = 64 >> ss_ver;
   81|  38.4k|            src += stripe_h * PXSTRIDE(src_stride);
  ------------------
  |  |   53|  38.4k|#define PXSTRIDE(x) (x)
  ------------------
   82|  38.4k|            dst += n_lines * PXSTRIDE(dst_stride);
  ------------------
  |  |   53|  38.4k|#define PXSTRIDE(x) (x)
  ------------------
   83|  38.4k|            if (n_lines == 3) {
  ------------------
  |  Branch (83:17): [True: 2.87k, False: 35.6k]
  ------------------
   84|  2.87k|                pixel_copy(dst, &dst[-PXSTRIDE(dst_stride)], dst_w);
  ------------------
  |  |   47|  2.87k|#define pixel_copy memcpy
  ------------------
                              pixel_copy(dst, &dst[-PXSTRIDE(dst_stride)], dst_w);
  ------------------
  |  |   53|  2.87k|#define PXSTRIDE(x) (x)
  ------------------
   85|  2.87k|                dst += PXSTRIDE(dst_stride);
  ------------------
  |  |   53|  2.87k|#define PXSTRIDE(x) (x)
  ------------------
   86|  2.87k|            }
   87|  38.4k|        }
   88|  1.53M|    } else {
   89|  3.38M|        while (row + stripe_h <= row_h) {
  ------------------
  |  Branch (89:16): [True: 1.85M, False: 1.53M]
  ------------------
   90|  1.85M|            const int n_lines = 4 - (row + stripe_h + 1 == h);
   91|  9.23M|            for (int i = 0; i < 4; i++) {
  ------------------
  |  Branch (91:29): [True: 7.38M, False: 1.85M]
  ------------------
   92|  7.38M|                pixel_copy(dst, i == n_lines ? &dst[-PXSTRIDE(dst_stride)] :
  ------------------
  |  |   47|  7.38M|#define pixel_copy memcpy
  ------------------
                              pixel_copy(dst, i == n_lines ? &dst[-PXSTRIDE(dst_stride)] :
  ------------------
  |  |   53|  8.52k|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (92:33): [True: 8.52k, False: 7.37M]
  ------------------
   93|  7.38M|                                               src, src_w);
   94|  7.38M|                dst += PXSTRIDE(dst_stride);
  ------------------
  |  |   53|  7.38M|#define PXSTRIDE(x) (x)
  ------------------
   95|  7.38M|                src += PXSTRIDE(src_stride);
  ------------------
  |  |   53|  7.38M|#define PXSTRIDE(x) (x)
  ------------------
   96|  7.38M|            }
   97|  1.85M|            row += stripe_h; // unmodified stripe_h for the 1st stripe
   98|  1.85M|            stripe_h = 64 >> ss_ver;
   99|  1.85M|            src += (stripe_h - 4) * PXSTRIDE(src_stride);
  ------------------
  |  |   53|  1.85M|#define PXSTRIDE(x) (x)
  ------------------
  100|  1.85M|        }
  101|  1.53M|    }
  102|  1.56M|}
lf_apply_tmpl.c:filter_plane_cols_y:
  184|  1.27M|{
  185|  1.27M|    const Dav1dDSPContext *const dsp = f->dsp;
  186|       |
  187|       |    // filter edges between columns (e.g. block1 | block2)
  188|  34.5M|    for (int x = 0; x < w; x++) {
  ------------------
  |  Branch (188:21): [True: 33.2M, False: 1.27M]
  ------------------
  189|  33.2M|        if (!have_left && !x) continue;
  ------------------
  |  Branch (189:13): [True: 31.6M, False: 1.68M]
  |  Branch (189:27): [True: 1.22M, False: 30.3M]
  ------------------
  190|  32.0M|        uint32_t hmask[4];
  191|  32.0M|        if (!starty4) {
  ------------------
  |  Branch (191:13): [True: 25.1M, False: 6.88M]
  ------------------
  192|  25.1M|            hmask[0] = mask[x][0][0];
  193|  25.1M|            hmask[1] = mask[x][1][0];
  194|  25.1M|            hmask[2] = mask[x][2][0];
  195|  25.1M|            if (endy4 > 16) {
  ------------------
  |  Branch (195:17): [True: 18.0M, False: 7.18M]
  ------------------
  196|  18.0M|                hmask[0] |= (unsigned) mask[x][0][1] << 16;
  197|  18.0M|                hmask[1] |= (unsigned) mask[x][1][1] << 16;
  198|  18.0M|                hmask[2] |= (unsigned) mask[x][2][1] << 16;
  199|  18.0M|            }
  200|  25.1M|        } else {
  201|  6.88M|            hmask[0] = mask[x][0][1];
  202|  6.88M|            hmask[1] = mask[x][1][1];
  203|  6.88M|            hmask[2] = mask[x][2][1];
  204|  6.88M|        }
  205|  32.0M|        hmask[3] = 0;
  206|  32.0M|        dsp->lf.loop_filter_sb[0][0](&dst[x * 4], ls, hmask,
  207|  32.0M|                                     (const uint8_t(*)[4]) &lvl[x][0], b4_stride,
  208|  32.0M|                                     &f->lf.lim_lut, endy4 - starty4 HIGHBD_CALL_SUFFIX);
  209|  32.0M|    }
  210|  1.27M|}
lf_apply_tmpl.c:filter_plane_cols_uv:
  251|   134k|{
  252|   134k|    const Dav1dDSPContext *const dsp = f->dsp;
  253|       |
  254|       |    // filter edges between columns (e.g. block1 | block2)
  255|  2.10M|    for (int x = 0; x < w; x++) {
  ------------------
  |  Branch (255:21): [True: 1.96M, False: 134k]
  ------------------
  256|  1.96M|        if (!have_left && !x) continue;
  ------------------
  |  Branch (256:13): [True: 1.29M, False: 671k]
  |  Branch (256:27): [True: 97.7k, False: 1.19M]
  ------------------
  257|  1.86M|        uint32_t hmask[3];
  258|  1.86M|        if (!starty4) {
  ------------------
  |  Branch (258:13): [True: 1.73M, False: 131k]
  ------------------
  259|  1.73M|            hmask[0] = mask[x][0][0];
  260|  1.73M|            hmask[1] = mask[x][1][0];
  261|  1.73M|            if (endy4 > (16 >> ss_ver)) {
  ------------------
  |  Branch (261:17): [True: 1.50M, False: 227k]
  ------------------
  262|  1.50M|                hmask[0] |= (unsigned) mask[x][0][1] << (16 >> ss_ver);
  263|  1.50M|                hmask[1] |= (unsigned) mask[x][1][1] << (16 >> ss_ver);
  264|  1.50M|            }
  265|  1.73M|        } else {
  266|   131k|            hmask[0] = mask[x][0][1];
  267|   131k|            hmask[1] = mask[x][1][1];
  268|   131k|        }
  269|  1.86M|        hmask[2] = 0;
  270|  1.86M|        dsp->lf.loop_filter_sb[1][0](&u[x * 4], ls, hmask,
  271|  1.86M|                                     (const uint8_t(*)[4]) &lvl[x][2], b4_stride,
  272|  1.86M|                                     &f->lf.lim_lut, endy4 - starty4 HIGHBD_CALL_SUFFIX);
  273|  1.86M|        dsp->lf.loop_filter_sb[1][0](&v[x * 4], ls, hmask,
  274|  1.86M|                                     (const uint8_t(*)[4]) &lvl[x][3], b4_stride,
  275|  1.86M|                                     &f->lf.lim_lut, endy4 - starty4 HIGHBD_CALL_SUFFIX);
  276|  1.86M|    }
  277|   134k|}
lf_apply_tmpl.c:filter_plane_rows_y:
  220|  1.27M|{
  221|  1.27M|    const Dav1dDSPContext *const dsp = f->dsp;
  222|       |
  223|       |    //                                 block1
  224|       |    // filter edges between rows (e.g. ------)
  225|       |    //                                 block2
  226|  33.3M|    for (int y = starty4; y < endy4;
  ------------------
  |  Branch (226:27): [True: 32.0M, False: 1.27M]
  ------------------
  227|  32.0M|         y++, dst += 4 * PXSTRIDE(ls), lvl += b4_stride)
  ------------------
  |  |   53|  32.0M|#define PXSTRIDE(x) (x)
  ------------------
  228|  32.0M|    {
  229|  32.0M|        if (!have_top && !y) continue;
  ------------------
  |  Branch (229:13): [True: 740k, False: 31.3M]
  |  Branch (229:26): [True: 33.5k, False: 706k]
  ------------------
  230|  32.0M|        const uint32_t vmask[4] = {
  231|  32.0M|            mask[y][0][0] | ((unsigned) mask[y][0][1] << 16),
  232|  32.0M|            mask[y][1][0] | ((unsigned) mask[y][1][1] << 16),
  233|  32.0M|            mask[y][2][0] | ((unsigned) mask[y][2][1] << 16),
  234|  32.0M|            0,
  235|  32.0M|        };
  236|  32.0M|        dsp->lf.loop_filter_sb[0][1](dst, ls, vmask,
  237|  32.0M|                                     (const uint8_t(*)[4]) &lvl[0][1], b4_stride,
  238|  32.0M|                                     &f->lf.lim_lut, w HIGHBD_CALL_SUFFIX);
  239|  32.0M|    }
  240|  1.27M|}
lf_apply_tmpl.c:filter_plane_rows_uv:
  288|   134k|{
  289|   134k|    const Dav1dDSPContext *const dsp = f->dsp;
  290|   134k|    ptrdiff_t off_l = 0;
  291|       |
  292|       |    //                                 block1
  293|       |    // filter edges between rows (e.g. ------)
  294|       |    //                                 block2
  295|  2.31M|    for (int y = starty4; y < endy4;
  ------------------
  |  Branch (295:27): [True: 2.18M, False: 134k]
  ------------------
  296|  2.18M|         y++, off_l += 4 * PXSTRIDE(ls), lvl += b4_stride)
  ------------------
  |  |   53|  2.18M|#define PXSTRIDE(x) (x)
  ------------------
  297|  2.18M|    {
  298|  2.18M|        if (!have_top && !y) continue;
  ------------------
  |  Branch (298:13): [True: 241k, False: 1.94M]
  |  Branch (298:26): [True: 15.5k, False: 226k]
  ------------------
  299|  2.16M|        const uint32_t vmask[3] = {
  300|  2.16M|            mask[y][0][0] | ((unsigned) mask[y][0][1] << (16 >> ss_hor)),
  301|  2.16M|            mask[y][1][0] | ((unsigned) mask[y][1][1] << (16 >> ss_hor)),
  302|  2.16M|            0,
  303|  2.16M|        };
  304|  2.16M|        dsp->lf.loop_filter_sb[1][1](&u[off_l], ls, vmask,
  305|  2.16M|                                     (const uint8_t(*)[4]) &lvl[0][2], b4_stride,
  306|  2.16M|                                     &f->lf.lim_lut, w HIGHBD_CALL_SUFFIX);
  307|  2.16M|        dsp->lf.loop_filter_sb[1][1](&v[off_l], ls, vmask,
  308|  2.16M|                                     (const uint8_t(*)[4]) &lvl[0][3], b4_stride,
  309|  2.16M|                                     &f->lf.lim_lut, w HIGHBD_CALL_SUFFIX);
  310|  2.16M|    }
  311|   134k|}
dav1d_copy_lpf_16bpc:
  106|   871k|{
  107|   871k|    const int have_tt = f->c->n_tc > 1;
  108|   871k|    const int resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
  109|   871k|    const int offset = 8 * !!sby;
  110|   871k|    const ptrdiff_t *const src_stride = f->cur.stride;
  111|   871k|    const ptrdiff_t *const lr_stride = f->sr_cur.p.stride;
  112|   871k|    const int tt_off = have_tt * sby * (4 << f->seq_hdr->sb128);
  113|   871k|    pixel *const dst[3] = {
  114|   871k|        f->lf.lr_lpf_line[0] + tt_off * PXSTRIDE(lr_stride[0]),
  115|   871k|        f->lf.lr_lpf_line[1] + tt_off * PXSTRIDE(lr_stride[1]),
  116|   871k|        f->lf.lr_lpf_line[2] + tt_off * PXSTRIDE(lr_stride[1])
  117|   871k|    };
  118|       |
  119|       |    // TODO Also check block level restore type to reduce copying.
  120|   871k|    const int restore_planes = f->lf.restore_planes;
  121|       |
  122|   871k|    if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_Y) {
  ------------------
  |  Branch (122:9): [True: 859k, False: 11.5k]
  |  Branch (122:29): [True: 10.8k, False: 683]
  ------------------
  123|   870k|        const int h = f->cur.p.h;
  124|   870k|        const int w = f->bw << 2;
  125|   870k|        const int row_h = imin((sby + 1) << (6 + f->seq_hdr->sb128), h - 1);
  126|   870k|        const int y_stripe = (sby << (6 + f->seq_hdr->sb128)) - offset;
  127|   870k|        if (restore_planes & LR_RESTORE_Y || !resize)
  ------------------
  |  Branch (127:13): [True: 41.4k, False: 828k]
  |  Branch (127:46): [True: 810k, False: 18.3k]
  ------------------
  128|   851k|            backup_lpf(f, dst[0], lr_stride[0],
  129|   851k|                       src[0] - offset * PXSTRIDE(src_stride[0]), src_stride[0],
  130|   851k|                       0, f->seq_hdr->sb128, y_stripe, row_h, w, h, 0, 1);
  131|   870k|        if (have_tt && resize) {
  ------------------
  |  Branch (131:13): [True: 869k, False: 211]
  |  Branch (131:24): [True: 28.1k, False: 841k]
  ------------------
  132|  28.1k|            const ptrdiff_t cdef_off_y = sby * 4 * PXSTRIDE(src_stride[0]);
  133|  28.1k|            backup_lpf(f, f->lf.cdef_lpf_line[0] + cdef_off_y, src_stride[0],
  134|  28.1k|                       src[0] - offset * PXSTRIDE(src_stride[0]), src_stride[0],
  135|  28.1k|                       0, f->seq_hdr->sb128, y_stripe, row_h, w, h, 0, 0);
  136|  28.1k|        }
  137|   870k|    }
  138|   871k|    if ((f->seq_hdr->cdef || restore_planes & (LR_RESTORE_U | LR_RESTORE_V)) &&
  ------------------
  |  Branch (138:10): [True: 859k, False: 11.5k]
  |  Branch (138:30): [True: 2.03k, False: 9.52k]
  ------------------
  139|   861k|        f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400)
  ------------------
  |  Branch (139:9): [True: 166k, False: 694k]
  ------------------
  140|   166k|    {
  141|   166k|        const int ss_ver = f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  142|   166k|        const int ss_hor = f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  143|   166k|        const int h = (f->cur.p.h + ss_ver) >> ss_ver;
  144|   166k|        const int w = f->bw << (2 - ss_hor);
  145|   166k|        const int row_h = imin((sby + 1) << ((6 - ss_ver) + f->seq_hdr->sb128), h - 1);
  146|   166k|        const int offset_uv = offset >> ss_ver;
  147|   166k|        const int y_stripe = (sby << ((6 - ss_ver) + f->seq_hdr->sb128)) - offset_uv;
  148|   166k|        const ptrdiff_t cdef_off_uv = sby * 4 * PXSTRIDE(src_stride[1]);
  149|   166k|        if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_U) {
  ------------------
  |  Branch (149:13): [True: 164k, False: 2.03k]
  |  Branch (149:33): [True: 1.45k, False: 574]
  ------------------
  150|   165k|            if (restore_planes & LR_RESTORE_U || !resize)
  ------------------
  |  Branch (150:17): [True: 6.60k, False: 159k]
  |  Branch (150:50): [True: 155k, False: 3.60k]
  ------------------
  151|   162k|                backup_lpf(f, dst[1], lr_stride[1],
  152|   162k|                           src[1] - offset_uv * PXSTRIDE(src_stride[1]),
  153|   162k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  154|   162k|                           row_h, w, h, ss_hor, 1);
  155|   165k|            if (have_tt && resize)
  ------------------
  |  Branch (155:17): [True: 165k, False: 18.4E]
  |  Branch (155:28): [True: 7.98k, False: 157k]
  ------------------
  156|  7.98k|                backup_lpf(f, f->lf.cdef_lpf_line[1] + cdef_off_uv, src_stride[1],
  157|  7.98k|                           src[1] - offset_uv * PXSTRIDE(src_stride[1]),
  158|  7.98k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  159|  7.98k|                           row_h, w, h, ss_hor, 0);
  160|   165k|        }
  161|   166k|        if (f->seq_hdr->cdef || restore_planes & LR_RESTORE_V) {
  ------------------
  |  Branch (161:13): [True: 164k, False: 2.03k]
  |  Branch (161:33): [True: 1.26k, False: 771]
  ------------------
  162|   165k|            if (restore_planes & LR_RESTORE_V || !resize)
  ------------------
  |  Branch (162:17): [True: 7.84k, False: 157k]
  |  Branch (162:50): [True: 155k, False: 2.53k]
  ------------------
  163|   163k|                backup_lpf(f, dst[2], lr_stride[1],
  164|   163k|                           src[2] - offset_uv * PXSTRIDE(src_stride[1]),
  165|   163k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  166|   163k|                           row_h, w, h, ss_hor, 1);
  167|   165k|            if (have_tt && resize)
  ------------------
  |  Branch (167:17): [True: 165k, False: 18.4E]
  |  Branch (167:28): [True: 7.80k, False: 157k]
  ------------------
  168|  7.80k|                backup_lpf(f, f->lf.cdef_lpf_line[2] + cdef_off_uv, src_stride[1],
  169|  7.80k|                           src[2] - offset_uv * PXSTRIDE(src_stride[1]),
  170|  7.80k|                           src_stride[1], ss_ver, f->seq_hdr->sb128, y_stripe,
  171|  7.80k|                           row_h, w, h, ss_hor, 0);
  172|   165k|        }
  173|   166k|    }
  174|   871k|}
dav1d_loopfilter_sbrow_cols_16bpc:
  316|   701k|{
  317|   701k|    int x, have_left;
  318|       |    // Don't filter outside the frame
  319|   701k|    const int is_sb64 = !f->seq_hdr->sb128;
  320|   701k|    const int starty4 = (sby & is_sb64) << 4;
  321|   701k|    const int sbsz = 32 >> is_sb64;
  322|   701k|    const int sbl2 = 5 - is_sb64;
  323|   701k|    const int halign = (f->bh + 31) & ~31;
  324|   701k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  325|   701k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  326|   701k|    const int vmask = 16 >> ss_ver, hmask = 16 >> ss_hor;
  327|   701k|    const unsigned vmax = 1U << vmask, hmax = 1U << hmask;
  328|   701k|    const unsigned endy4 = starty4 + imin(f->h4 - sby * sbsz, sbsz);
  329|   701k|    const unsigned uv_endy4 = (endy4 + ss_ver) >> ss_ver;
  330|       |
  331|       |    // fix lpf strength at tile col boundaries
  332|   701k|    const uint8_t *lpf_y = &f->lf.tx_lpf_right_edge[0][sby << sbl2];
  333|   701k|    const uint8_t *lpf_uv = &f->lf.tx_lpf_right_edge[1][sby << (sbl2 - ss_ver)];
  334|   709k|    for (int tile_col = 1;; tile_col++) {
  335|   709k|        x = f->frame_hdr->tiling.col_start_sb[tile_col];
  336|   709k|        if ((x << sbl2) >= f->bw) break;
  ------------------
  |  Branch (336:13): [True: 701k, False: 8.14k]
  ------------------
  337|  8.14k|        const int bx4 = x & is_sb64 ? 16 : 0, cbx4 = bx4 >> ss_hor;
  ------------------
  |  Branch (337:25): [True: 6.33k, False: 1.81k]
  ------------------
  338|  8.14k|        x >>= is_sb64;
  339|       |
  340|  8.14k|        uint16_t (*const y_hmask)[2] = lflvl[x].filter_y[0][bx4];
  341|   162k|        for (unsigned y = starty4, mask = 1 << y; y < endy4; y++, mask <<= 1) {
  ------------------
  |  Branch (341:51): [True: 154k, False: 8.14k]
  ------------------
  342|   154k|            const int sidx = mask >= 0x10000U;
  343|   154k|            const unsigned smask = mask >> (sidx << 4);
  344|   154k|            const int idx = 2 * !!(y_hmask[2][sidx] & smask) +
  345|   154k|                                !!(y_hmask[1][sidx] & smask);
  346|   154k|            y_hmask[2][sidx] &= ~smask;
  347|   154k|            y_hmask[1][sidx] &= ~smask;
  348|   154k|            y_hmask[0][sidx] &= ~smask;
  349|   154k|            y_hmask[imin(idx, lpf_y[y - starty4])][sidx] |= smask;
  350|   154k|        }
  351|       |
  352|  8.14k|        if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (352:13): [True: 1.06k, False: 7.08k]
  ------------------
  353|  1.06k|            uint16_t (*const uv_hmask)[2] = lflvl[x].filter_uv[0][cbx4];
  354|  28.2k|            for (unsigned y = starty4 >> ss_ver, uv_mask = 1 << y; y < uv_endy4;
  ------------------
  |  Branch (354:68): [True: 27.2k, False: 1.06k]
  ------------------
  355|  27.2k|                 y++, uv_mask <<= 1)
  356|  27.2k|            {
  357|  27.2k|                const int sidx = uv_mask >= vmax;
  358|  27.2k|                const unsigned smask = uv_mask >> (sidx << (4 - ss_ver));
  359|  27.2k|                const int idx = !!(uv_hmask[1][sidx] & smask);
  360|  27.2k|                uv_hmask[1][sidx] &= ~smask;
  361|  27.2k|                uv_hmask[0][sidx] &= ~smask;
  362|  27.2k|                uv_hmask[imin(idx, lpf_uv[y - (starty4 >> ss_ver)])][sidx] |= smask;
  363|  27.2k|            }
  364|  1.06k|        }
  365|  8.14k|        lpf_y  += halign;
  366|  8.14k|        lpf_uv += halign >> ss_ver;
  367|  8.14k|    }
  368|       |
  369|       |    // fix lpf strength at tile row boundaries
  370|   701k|    if (start_of_tile_row) {
  ------------------
  |  Branch (370:9): [True: 1.46k, False: 700k]
  ------------------
  371|  1.46k|        const BlockContext *a;
  372|  1.46k|        for (x = 0, a = &f->a[f->sb128w * (start_of_tile_row - 1)];
  373|  5.19k|             x < f->sb128w; x++, a++)
  ------------------
  |  Branch (373:14): [True: 3.73k, False: 1.46k]
  ------------------
  374|  3.73k|        {
  375|  3.73k|            uint16_t (*const y_vmask)[2] = lflvl[x].filter_y[1][starty4];
  376|  3.73k|            const unsigned w = imin(32, f->w4 - (x << 5));
  377|   113k|            for (unsigned mask = 1, i = 0; i < w; mask <<= 1, i++) {
  ------------------
  |  Branch (377:44): [True: 109k, False: 3.73k]
  ------------------
  378|   109k|                const int sidx = mask >= 0x10000U;
  379|   109k|                const unsigned smask = mask >> (sidx << 4);
  380|   109k|                const int idx = 2 * !!(y_vmask[2][sidx] & smask) +
  381|   109k|                                    !!(y_vmask[1][sidx] & smask);
  382|   109k|                y_vmask[2][sidx] &= ~smask;
  383|   109k|                y_vmask[1][sidx] &= ~smask;
  384|   109k|                y_vmask[0][sidx] &= ~smask;
  385|   109k|                y_vmask[imin(idx, a->tx_lpf_y[i])][sidx] |= smask;
  386|   109k|            }
  387|       |
  388|  3.73k|            if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (388:17): [True: 2.39k, False: 1.33k]
  ------------------
  389|  2.39k|                const unsigned cw = (w + ss_hor) >> ss_hor;
  390|  2.39k|                uint16_t (*const uv_vmask)[2] = lflvl[x].filter_uv[1][starty4 >> ss_ver];
  391|  49.2k|                for (unsigned uv_mask = 1, i = 0; i < cw; uv_mask <<= 1, i++) {
  ------------------
  |  Branch (391:51): [True: 46.8k, False: 2.39k]
  ------------------
  392|  46.8k|                    const int sidx = uv_mask >= hmax;
  393|  46.8k|                    const unsigned smask = uv_mask >> (sidx << (4 - ss_hor));
  394|  46.8k|                    const int idx = !!(uv_vmask[1][sidx] & smask);
  395|  46.8k|                    uv_vmask[1][sidx] &= ~smask;
  396|  46.8k|                    uv_vmask[0][sidx] &= ~smask;
  397|  46.8k|                    uv_vmask[imin(idx, a->tx_lpf_uv[i])][sidx] |= smask;
  398|  46.8k|                }
  399|  2.39k|            }
  400|  3.73k|        }
  401|  1.46k|    }
  402|       |
  403|   701k|    pixel *ptr;
  404|   701k|    uint8_t (*level_ptr)[4] = f->lf.level + f->b4_stride * sby * sbsz;
  405|  1.42M|    for (ptr = p[0], have_left = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (405:44): [True: 724k, False: 701k]
  ------------------
  406|   724k|         x++, have_left = 1, ptr += 128, level_ptr += 32)
  407|   724k|    {
  408|   724k|        filter_plane_cols_y(f, have_left, level_ptr, f->b4_stride,
  409|   724k|                            lflvl[x].filter_y[0], ptr, f->cur.stride[0],
  410|   724k|                            imin(32, f->w4 - x * 32), starty4, endy4);
  411|   724k|    }
  412|       |
  413|   701k|    if (!f->frame_hdr->loopfilter.level_u && !f->frame_hdr->loopfilter.level_v)
  ------------------
  |  Branch (413:9): [True: 640k, False: 61.6k]
  |  Branch (413:46): [True: 637k, False: 2.07k]
  ------------------
  414|   637k|        return;
  415|       |
  416|  63.7k|    ptrdiff_t uv_off;
  417|  63.7k|    level_ptr = f->lf.level + f->b4_stride * (sby * sbsz >> ss_ver);
  418|   139k|    for (uv_off = 0, have_left = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (418:44): [True: 75.6k, False: 63.7k]
  ------------------
  419|  75.6k|         x++, have_left = 1, uv_off += 128 >> ss_hor, level_ptr += 32 >> ss_hor)
  420|  75.6k|    {
  421|  75.6k|        filter_plane_cols_uv(f, have_left, level_ptr, f->b4_stride,
  422|  75.6k|                             lflvl[x].filter_uv[0],
  423|  75.6k|                             &p[1][uv_off], &p[2][uv_off], f->cur.stride[1],
  424|  75.6k|                             (imin(32, f->w4 - x * 32) + ss_hor) >> ss_hor,
  425|  75.6k|                             starty4 >> ss_ver, uv_endy4, ss_ver);
  426|  75.6k|    }
  427|  63.7k|}
dav1d_loopfilter_sbrow_rows_16bpc:
  432|   702k|{
  433|   702k|    int x;
  434|       |    // Don't filter outside the frame
  435|   702k|    const int have_top = sby > 0;
  436|   702k|    const int is_sb64 = !f->seq_hdr->sb128;
  437|   702k|    const int starty4 = (sby & is_sb64) << 4;
  438|   702k|    const int sbsz = 32 >> is_sb64;
  439|   702k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  440|   702k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  441|   702k|    const unsigned endy4 = starty4 + imin(f->h4 - sby * sbsz, sbsz);
  442|   702k|    const unsigned uv_endy4 = (endy4 + ss_ver) >> ss_ver;
  443|       |
  444|   702k|    pixel *ptr;
  445|   702k|    uint8_t (*level_ptr)[4] = f->lf.level + f->b4_stride * sby * sbsz;
  446|  1.42M|    for (ptr = p[0], x = 0; x < f->sb128w; x++, ptr += 128, level_ptr += 32) {
  ------------------
  |  Branch (446:29): [True: 725k, False: 702k]
  ------------------
  447|   725k|        filter_plane_rows_y(f, have_top, level_ptr, f->b4_stride,
  448|   725k|                            lflvl[x].filter_y[1], ptr, f->cur.stride[0],
  449|   725k|                            imin(32, f->w4 - x * 32), starty4, endy4);
  450|   725k|    }
  451|       |
  452|   702k|    if (!f->frame_hdr->loopfilter.level_u && !f->frame_hdr->loopfilter.level_v)
  ------------------
  |  Branch (452:9): [True: 639k, False: 62.1k]
  |  Branch (452:46): [True: 637k, False: 2.07k]
  ------------------
  453|   637k|        return;
  454|       |
  455|  64.2k|    ptrdiff_t uv_off;
  456|  64.2k|    level_ptr = f->lf.level + f->b4_stride * (sby * sbsz >> ss_ver);
  457|   139k|    for (uv_off = 0, x = 0; x < f->sb128w;
  ------------------
  |  Branch (457:29): [True: 75.6k, False: 64.2k]
  ------------------
  458|  75.6k|         x++, uv_off += 128 >> ss_hor, level_ptr += 32 >> ss_hor)
  459|  75.6k|    {
  460|  75.6k|        filter_plane_rows_uv(f, have_top, level_ptr, f->b4_stride,
  461|  75.6k|                             lflvl[x].filter_uv[1],
  462|  75.6k|                             &p[1][uv_off], &p[2][uv_off], f->cur.stride[1],
  463|  75.6k|                             (imin(32, f->w4 - x * 32) + ss_hor) >> ss_hor,
  464|  75.6k|                             starty4 >> ss_ver, uv_endy4, ss_hor);
  465|  75.6k|    }
  466|  64.2k|}

dav1d_create_lf_mask_intra:
  271|  1.53M|{
  272|  1.53M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  273|  1.53M|    const int bw4 = imin(iw - bx, b_dim[0]);
  274|  1.53M|    const int bh4 = imin(ih - by, b_dim[1]);
  275|  1.53M|    const int bx4 = bx & 31;
  276|  1.53M|    const int by4 = by & 31;
  277|  1.53M|    assert(bw4 >= 0 && bh4 >= 0);
  ------------------
  |  Branch (277:5): [True: 1.53M, False: 4.72k]
  |  Branch (277:5): [True: 1.53M, False: 18.4E]
  ------------------
  278|       |
  279|  1.53M|    if (bw4 && bh4) {
  ------------------
  |  Branch (279:9): [True: 1.52M, False: 3.68k]
  |  Branch (279:16): [True: 1.50M, False: 28.4k]
  ------------------
  280|  1.50M|        uint8_t (*level_cache_ptr)[4] = level_cache + by * b4_stride + bx;
  281|  8.47M|        for (int y = 0; y < bh4; y++) {
  ------------------
  |  Branch (281:25): [True: 6.97M, False: 1.50M]
  ------------------
  282|  66.4M|            for (int x = 0; x < bw4; x++) {
  ------------------
  |  Branch (282:29): [True: 59.4M, False: 6.97M]
  ------------------
  283|  59.4M|                level_cache_ptr[x][0] = filter_level[0][0][0];
  284|  59.4M|                level_cache_ptr[x][1] = filter_level[1][0][0];
  285|  59.4M|            }
  286|  6.97M|            level_cache_ptr += b4_stride;
  287|  6.97M|        }
  288|       |
  289|  1.50M|        mask_edges_intra(lflvl->filter_y, by4, bx4, bw4, bh4, ytx, ay, ly);
  290|  1.50M|    }
  291|       |
  292|  1.53M|    if (!auv) return;
  ------------------
  |  Branch (292:9): [True: 453k, False: 1.07M]
  ------------------
  293|       |
  294|  1.07M|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
  295|  1.07M|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
  296|  1.07M|    const int cbw4 = imin(((iw + ss_hor) >> ss_hor) - (bx >> ss_hor),
  297|  1.07M|                          (b_dim[0] + ss_hor) >> ss_hor);
  298|  1.07M|    const int cbh4 = imin(((ih + ss_ver) >> ss_ver) - (by >> ss_ver),
  299|  1.07M|                          (b_dim[1] + ss_ver) >> ss_ver);
  300|  1.07M|    assert(cbw4 >= 0 && cbh4 >= 0);
  ------------------
  |  Branch (300:5): [True: 1.08M, False: 18.4E]
  |  Branch (300:5): [True: 1.08M, False: 18.4E]
  ------------------
  301|       |
  302|  1.08M|    if (!cbw4 || !cbh4) return;
  ------------------
  |  Branch (302:9): [True: 1.92k, False: 1.08M]
  |  Branch (302:18): [True: 8.92k, False: 1.07M]
  ------------------
  303|       |
  304|  1.07M|    const int cbx4 = bx4 >> ss_hor;
  305|  1.07M|    const int cby4 = by4 >> ss_ver;
  306|       |
  307|  1.07M|    uint8_t (*level_cache_ptr)[4] =
  308|  1.07M|        level_cache + (by >> ss_ver) * b4_stride + (bx >> ss_hor);
  309|  4.56M|    for (int y = 0; y < cbh4; y++) {
  ------------------
  |  Branch (309:21): [True: 3.48M, False: 1.07M]
  ------------------
  310|  24.4M|        for (int x = 0; x < cbw4; x++) {
  ------------------
  |  Branch (310:25): [True: 20.9M, False: 3.48M]
  ------------------
  311|  20.9M|            level_cache_ptr[x][2] = filter_level[2][0][0];
  312|  20.9M|            level_cache_ptr[x][3] = filter_level[3][0][0];
  313|  20.9M|        }
  314|  3.48M|        level_cache_ptr += b4_stride;
  315|  3.48M|    }
  316|       |
  317|  1.07M|    mask_edges_chroma(lflvl->filter_uv, cby4, cbx4, cbw4, cbh4, 0, uvtx,
  318|  1.07M|                      auv, luv, ss_hor, ss_ver);
  319|  1.07M|}
dav1d_create_lf_mask_inter:
  334|  3.63M|{
  335|  3.63M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  336|  3.63M|    const int bw4 = imin(iw - bx, b_dim[0]);
  337|  3.63M|    const int bh4 = imin(ih - by, b_dim[1]);
  338|  3.63M|    const int bx4 = bx & 31;
  339|  3.63M|    const int by4 = by & 31;
  340|  3.63M|    assert(bw4 >= 0 && bh4 >= 0);
  ------------------
  |  Branch (340:5): [True: 3.63M, False: 7.49k]
  |  Branch (340:5): [True: 3.63M, False: 18.4E]
  ------------------
  341|       |
  342|  3.63M|    if (bw4 && bh4) {
  ------------------
  |  Branch (342:9): [True: 3.62M, False: 3.85k]
  |  Branch (342:16): [True: 3.62M, False: 7.83k]
  ------------------
  343|  3.62M|        uint8_t (*level_cache_ptr)[4] = level_cache + by * b4_stride + bx;
  344|  49.6M|        for (int y = 0; y < bh4; y++) {
  ------------------
  |  Branch (344:25): [True: 46.0M, False: 3.62M]
  ------------------
  345|   866M|            for (int x = 0; x < bw4; x++) {
  ------------------
  |  Branch (345:29): [True: 820M, False: 46.0M]
  ------------------
  346|   820M|                level_cache_ptr[x][0] = filter_level[0][0][0];
  347|   820M|                level_cache_ptr[x][1] = filter_level[1][0][0];
  348|   820M|            }
  349|  46.0M|            level_cache_ptr += b4_stride;
  350|  46.0M|        }
  351|       |
  352|  3.62M|        mask_edges_inter(lflvl->filter_y, by4, bx4, bw4, bh4, skip,
  353|  3.62M|                         max_ytx, tx_masks, ay, ly);
  354|  3.62M|    }
  355|       |
  356|  3.63M|    if (!auv) return;
  ------------------
  |  Branch (356:9): [True: 2.51M, False: 1.11M]
  ------------------
  357|       |
  358|  1.11M|    const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
  359|  1.11M|    const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
  360|  1.11M|    const int cbw4 = imin(((iw + ss_hor) >> ss_hor) - (bx >> ss_hor),
  361|  1.11M|                          (b_dim[0] + ss_hor) >> ss_hor);
  362|  1.11M|    const int cbh4 = imin(((ih + ss_ver) >> ss_ver) - (by >> ss_ver),
  363|  1.11M|                          (b_dim[1] + ss_ver) >> ss_ver);
  364|  1.11M|    assert(cbw4 >= 0 && cbh4 >= 0);
  ------------------
  |  Branch (364:5): [True: 1.11M, False: 18.4E]
  |  Branch (364:5): [True: 1.11M, False: 104]
  ------------------
  365|       |
  366|  1.11M|    if (!cbw4 || !cbh4) return;
  ------------------
  |  Branch (366:9): [True: 831, False: 1.11M]
  |  Branch (366:18): [True: 1.66k, False: 1.11M]
  ------------------
  367|       |
  368|  1.11M|    const int cbx4 = bx4 >> ss_hor;
  369|  1.11M|    const int cby4 = by4 >> ss_ver;
  370|       |
  371|  1.11M|    uint8_t (*level_cache_ptr)[4] =
  372|  1.11M|        level_cache + (by >> ss_ver) * b4_stride + (bx >> ss_hor);
  373|  5.46M|    for (int y = 0; y < cbh4; y++) {
  ------------------
  |  Branch (373:21): [True: 4.35M, False: 1.11M]
  ------------------
  374|  38.5M|        for (int x = 0; x < cbw4; x++) {
  ------------------
  |  Branch (374:25): [True: 34.1M, False: 4.35M]
  ------------------
  375|  34.1M|            level_cache_ptr[x][2] = filter_level[2][0][0];
  376|  34.1M|            level_cache_ptr[x][3] = filter_level[3][0][0];
  377|  34.1M|        }
  378|  4.35M|        level_cache_ptr += b4_stride;
  379|  4.35M|    }
  380|       |
  381|  1.11M|    mask_edges_chroma(lflvl->filter_uv, cby4, cbx4, cbw4, cbh4, skip, uvtx,
  382|  1.11M|                      auv, luv, ss_hor, ss_ver);
  383|  1.11M|}
dav1d_calc_eih:
  385|  34.1k|void dav1d_calc_eih(Av1FilterLUT *const lim_lut, const int filter_sharpness) {
  386|       |    // set E/I/H values from loopfilter level
  387|  34.1k|    const int sharp = filter_sharpness;
  388|  2.21M|    for (int level = 0; level < 64; level++) {
  ------------------
  |  Branch (388:25): [True: 2.17M, False: 34.1k]
  ------------------
  389|  2.17M|        int limit = level;
  390|       |
  391|  2.17M|        if (sharp > 0) {
  ------------------
  |  Branch (391:13): [True: 1.11M, False: 1.06M]
  ------------------
  392|  1.11M|            limit >>= (sharp + 3) >> 2;
  393|  1.11M|            limit = imin(limit, 9 - sharp);
  394|  1.11M|        }
  395|  2.17M|        limit = imax(limit, 1);
  396|       |
  397|  2.17M|        lim_lut->i[level] = limit;
  398|  2.17M|        lim_lut->e[level] = 2 * (level + 2) + limit;
  399|  2.17M|    }
  400|  34.1k|    lim_lut->sharp[0] = (sharp + 3) >> 2;
  401|  34.1k|    lim_lut->sharp[1] = sharp ? 9 - sharp : 0xff;
  ------------------
  |  Branch (401:25): [True: 17.3k, False: 16.7k]
  ------------------
  402|  34.1k|}
dav1d_calc_lf_values:
  441|   371k|{
  442|   371k|    const int n_seg = hdr->segmentation.enabled ? 8 : 1;
  ------------------
  |  Branch (442:23): [True: 25.2k, False: 346k]
  ------------------
  443|       |
  444|   371k|    if (!hdr->loopfilter.level_y[0] && !hdr->loopfilter.level_y[1]) {
  ------------------
  |  Branch (444:9): [True: 282k, False: 89.3k]
  |  Branch (444:40): [True: 272k, False: 9.52k]
  ------------------
  445|   272k|        memset(lflvl_values, 0, sizeof(*lflvl_values) * n_seg);
  446|   272k|        return;
  447|   272k|    }
  448|       |
  449|  98.8k|    const Dav1dLoopfilterModeRefDeltas *const mr_deltas =
  450|  98.8k|        hdr->loopfilter.mode_ref_delta_enabled ?
  ------------------
  |  Branch (450:9): [True: 67.1k, False: 31.7k]
  ------------------
  451|  98.8k|        &hdr->loopfilter.mode_ref_deltas : NULL;
  452|   349k|    for (int s = 0; s < n_seg; s++) {
  ------------------
  |  Branch (452:21): [True: 250k, False: 98.8k]
  ------------------
  453|   250k|        const Dav1dSegmentationData *const segd =
  454|   250k|            hdr->segmentation.enabled ? &hdr->segmentation.seg_data.d[s] : NULL;
  ------------------
  |  Branch (454:13): [True: 173k, False: 77.1k]
  ------------------
  455|       |
  456|   250k|        calc_lf_value(lflvl_values[s][0], hdr->loopfilter.level_y[0],
  457|   250k|                      lf_delta[0], segd ? segd->delta_lf_y_v : 0, mr_deltas);
  ------------------
  |  Branch (457:36): [True: 173k, False: 77.1k]
  ------------------
  458|   250k|        calc_lf_value(lflvl_values[s][1], hdr->loopfilter.level_y[1],
  459|   250k|                      lf_delta[hdr->delta.lf.multi ? 1 : 0],
  ------------------
  |  Branch (459:32): [True: 89.5k, False: 161k]
  ------------------
  460|   250k|                      segd ? segd->delta_lf_y_h : 0, mr_deltas);
  ------------------
  |  Branch (460:23): [True: 173k, False: 77.2k]
  ------------------
  461|   250k|        calc_lf_value_chroma(lflvl_values[s][2], hdr->loopfilter.level_u,
  462|   250k|                             lf_delta[hdr->delta.lf.multi ? 2 : 0],
  ------------------
  |  Branch (462:39): [True: 89.5k, False: 161k]
  ------------------
  463|   250k|                             segd ? segd->delta_lf_u : 0, mr_deltas);
  ------------------
  |  Branch (463:30): [True: 173k, False: 77.1k]
  ------------------
  464|   250k|        calc_lf_value_chroma(lflvl_values[s][3], hdr->loopfilter.level_v,
  465|   250k|                             lf_delta[hdr->delta.lf.multi ? 3 : 0],
  ------------------
  |  Branch (465:39): [True: 89.5k, False: 161k]
  ------------------
  466|   250k|                             segd ? segd->delta_lf_v : 0, mr_deltas);
  ------------------
  |  Branch (466:30): [True: 173k, False: 77.1k]
  ------------------
  467|   250k|    }
  468|  98.8k|}
lf_mask.c:mask_edges_intra:
  152|  1.50M|{
  153|  1.50M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[tx];
  154|  1.50M|    const int twl4 = t_dim->lw, thl4 = t_dim->lh;
  155|  1.50M|    const int twl4c = imin(2, twl4), thl4c = imin(2, thl4);
  156|  1.50M|    int y, x;
  157|       |
  158|       |    // left block edge
  159|  1.50M|    unsigned mask = 1U << by4;
  160|  8.51M|    for (y = 0; y < h4; y++, mask <<= 1) {
  ------------------
  |  Branch (160:17): [True: 7.01M, False: 1.50M]
  ------------------
  161|  7.01M|        const int sidx = mask >= 0x10000;
  162|  7.01M|        const unsigned smask = mask >> (sidx << 4);
  163|  7.01M|        masks[0][bx4][imin(twl4c, l[y])][sidx] |= smask;
  164|  7.01M|    }
  165|       |
  166|       |    // top block edge
  167|  8.43M|    for (x = 0, mask = 1U << bx4; x < w4; x++, mask <<= 1) {
  ------------------
  |  Branch (167:35): [True: 6.93M, False: 1.50M]
  ------------------
  168|  6.93M|        const int sidx = mask >= 0x10000;
  169|  6.93M|        const unsigned smask = mask >> (sidx << 4);
  170|  6.93M|        masks[1][by4][imin(thl4c, a[x])][sidx] |= smask;
  171|  6.93M|    }
  172|       |
  173|       |    // inner (tx) left|right edges
  174|  1.50M|    const int hstep = t_dim->w;
  175|  1.50M|    unsigned t = 1U << by4;
  176|  1.50M|    unsigned inner = (unsigned) ((((uint64_t) t) << h4) - t);
  177|  1.50M|    unsigned inner1 = inner & 0xffff, inner2 = inner >> 16;
  178|  1.94M|    for (x = hstep; x < w4; x += hstep) {
  ------------------
  |  Branch (178:21): [True: 441k, False: 1.50M]
  ------------------
  179|   441k|        if (inner1) masks[0][bx4 + x][twl4c][0] |= inner1;
  ------------------
  |  Branch (179:13): [True: 340k, False: 100k]
  ------------------
  180|   441k|        if (inner2) masks[0][bx4 + x][twl4c][1] |= inner2;
  ------------------
  |  Branch (180:13): [True: 247k, False: 193k]
  ------------------
  181|   441k|    }
  182|       |
  183|       |    //            top
  184|       |    // inner (tx) --- edges
  185|       |    //           bottom
  186|  1.50M|    const int vstep = t_dim->h;
  187|  1.50M|    t = 1U << bx4;
  188|  1.50M|    inner = (unsigned) ((((uint64_t) t) << w4) - t);
  189|  1.50M|    inner1 = inner & 0xffff;
  190|  1.50M|    inner2 = inner >> 16;
  191|  2.02M|    for (y = vstep; y < h4; y += vstep) {
  ------------------
  |  Branch (191:21): [True: 520k, False: 1.50M]
  ------------------
  192|   520k|        if (inner1) masks[1][by4 + y][thl4c][0] |= inner1;
  ------------------
  |  Branch (192:13): [True: 386k, False: 133k]
  ------------------
  193|   520k|        if (inner2) masks[1][by4 + y][thl4c][1] |= inner2;
  ------------------
  |  Branch (193:13): [True: 360k, False: 159k]
  ------------------
  194|   520k|    }
  195|       |
  196|  1.50M|    dav1d_memset_likely_pow2(a, thl4c, w4);
  197|  1.50M|    dav1d_memset_likely_pow2(l, twl4c, h4);
  198|  1.50M|}
lf_mask.c:mask_edges_chroma:
  207|  2.18M|{
  208|  2.18M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[tx];
  209|  2.18M|    const int twl4 = t_dim->lw, thl4 = t_dim->lh;
  210|  2.18M|    const int twl4c = !!twl4, thl4c = !!thl4;
  211|  2.18M|    int y, x;
  212|  2.18M|    const int vbits = 4 - ss_ver, hbits = 4 - ss_hor;
  213|  2.18M|    const int vmask = 16 >> ss_ver, hmask = 16 >> ss_hor;
  214|  2.18M|    const unsigned vmax = 1 << vmask, hmax = 1 << hmask;
  215|       |
  216|       |    // left block edge
  217|  2.18M|    unsigned mask = 1U << cby4;
  218|  10.0M|    for (y = 0; y < ch4; y++, mask <<= 1) {
  ------------------
  |  Branch (218:17): [True: 7.86M, False: 2.18M]
  ------------------
  219|  7.86M|        const int sidx = mask >= vmax;
  220|  7.86M|        const unsigned smask = mask >> (sidx << vbits);
  221|  7.86M|        masks[0][cbx4][imin(twl4c, l[y])][sidx] |= smask;
  222|  7.86M|    }
  223|       |
  224|       |    // top block edge
  225|  9.27M|    for (x = 0, mask = 1U << cbx4; x < cw4; x++, mask <<= 1) {
  ------------------
  |  Branch (225:36): [True: 7.08M, False: 2.18M]
  ------------------
  226|  7.08M|        const int sidx = mask >= hmax;
  227|  7.08M|        const unsigned smask = mask >> (sidx << hbits);
  228|  7.08M|        masks[1][cby4][imin(thl4c, a[x])][sidx] |= smask;
  229|  7.08M|    }
  230|       |
  231|  2.18M|    if (!skip_inter) {
  ------------------
  |  Branch (231:9): [True: 1.64M, False: 545k]
  ------------------
  232|       |        // inner (tx) left|right edges
  233|  1.64M|        const int hstep = t_dim->w;
  234|  1.64M|        unsigned t = 1U << cby4;
  235|  1.64M|        unsigned inner = (unsigned) ((((uint64_t) t) << ch4) - t);
  236|  1.64M|        unsigned inner1 = inner & ((1 << vmask) - 1), inner2 = inner >> vmask;
  237|  1.78M|        for (x = hstep; x < cw4; x += hstep) {
  ------------------
  |  Branch (237:25): [True: 144k, False: 1.64M]
  ------------------
  238|   144k|            if (inner1) masks[0][cbx4 + x][twl4c][0] |= inner1;
  ------------------
  |  Branch (238:17): [True: 127k, False: 17.2k]
  ------------------
  239|   144k|            if (inner2) masks[0][cbx4 + x][twl4c][1] |= inner2;
  ------------------
  |  Branch (239:17): [True: 116k, False: 27.8k]
  ------------------
  240|   144k|        }
  241|       |
  242|       |        //            top
  243|       |        // inner (tx) --- edges
  244|       |        //           bottom
  245|  1.64M|        const int vstep = t_dim->h;
  246|  1.64M|        t = 1U << cbx4;
  247|  1.64M|        inner = (unsigned) ((((uint64_t) t) << cw4) - t);
  248|  1.64M|        inner1 = inner & ((1 << hmask) - 1), inner2 = inner >> hmask;
  249|  1.92M|        for (y = vstep; y < ch4; y += vstep) {
  ------------------
  |  Branch (249:25): [True: 280k, False: 1.64M]
  ------------------
  250|   280k|            if (inner1) masks[1][cby4 + y][thl4c][0] |= inner1;
  ------------------
  |  Branch (250:17): [True: 253k, False: 26.9k]
  ------------------
  251|   280k|            if (inner2) masks[1][cby4 + y][thl4c][1] |= inner2;
  ------------------
  |  Branch (251:17): [True: 236k, False: 43.9k]
  ------------------
  252|   280k|        }
  253|  1.64M|    }
  254|       |
  255|  2.18M|    dav1d_memset_likely_pow2(a, thl4c, cw4);
  256|  2.18M|    dav1d_memset_likely_pow2(l, twl4c, ch4);
  257|  2.18M|}
lf_mask.c:mask_edges_inter:
   85|  3.62M|{
   86|  3.62M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[max_tx];
   87|  3.62M|    int y, x;
   88|       |
   89|  3.62M|    ALIGN_STK_16(uint8_t, txa, 2 /* edge */, [2 /* txsz, step */][32 /* y */][32 /* x */]);
  ------------------
  |  |  100|  3.62M|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  3.62M|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
   90|  8.30M|    for (int y_off = 0, y = 0; y < h4; y += t_dim->h, y_off++)
  ------------------
  |  Branch (90:32): [True: 4.67M, False: 3.62M]
  ------------------
   91|  16.4M|        for (int x_off = 0, x = 0; x < w4; x += t_dim->w, x_off++)
  ------------------
  |  Branch (91:36): [True: 11.7M, False: 4.67M]
  ------------------
   92|  11.7M|            decomp_tx((uint8_t(*)[2][32][32]) &txa[0][0][y][x],
   93|  11.7M|                      max_tx, 0, y_off, x_off, tx_masks);
   94|       |
   95|       |    // left block edge
   96|  3.62M|    unsigned mask = 1U << by4;
   97|  50.1M|    for (y = 0; y < h4; y++, mask <<= 1) {
  ------------------
  |  Branch (97:17): [True: 46.5M, False: 3.62M]
  ------------------
   98|  46.5M|        const int sidx = mask >= 0x10000;
   99|  46.5M|        const unsigned smask = mask >> (sidx << 4);
  100|  46.5M|        masks[0][bx4][imin(txa[0][0][y][0], l[y])][sidx] |= smask;
  101|  46.5M|    }
  102|       |
  103|       |    // top block edge
  104|  42.9M|    for (x = 0, mask = 1U << bx4; x < w4; x++, mask <<= 1) {
  ------------------
  |  Branch (104:35): [True: 39.3M, False: 3.62M]
  ------------------
  105|  39.3M|        const int sidx = mask >= 0x10000;
  106|  39.3M|        const unsigned smask = mask >> (sidx << 4);
  107|  39.3M|        masks[1][by4][imin(txa[1][0][0][x], a[x])][sidx] |= smask;
  108|  39.3M|    }
  109|       |
  110|  3.62M|    if (!skip) {
  ------------------
  |  Branch (110:9): [True: 1.09M, False: 2.52M]
  ------------------
  111|       |        // inner (tx) left|right edges
  112|  5.41M|        for (y = 0, mask = 1U << by4; y < h4; y++, mask <<= 1) {
  ------------------
  |  Branch (112:39): [True: 4.32M, False: 1.09M]
  ------------------
  113|  4.32M|            const int sidx = mask >= 0x10000U;
  114|  4.32M|            const unsigned smask = mask >> (sidx << 4);
  115|  4.32M|            int ltx = txa[0][0][y][0];
  116|  4.32M|            int step = txa[0][1][y][0];
  117|  5.30M|            for (x = step; x < w4; x += step) {
  ------------------
  |  Branch (117:28): [True: 983k, False: 4.32M]
  ------------------
  118|   983k|                const int rtx = txa[0][0][y][x];
  119|   983k|                masks[0][bx4 + x][imin(rtx, ltx)][sidx] |= smask;
  120|   983k|                ltx = rtx;
  121|   983k|                step = txa[0][1][y][x];
  122|   983k|            }
  123|  4.32M|        }
  124|       |
  125|       |        //            top
  126|       |        // inner (tx) --- edges
  127|       |        //           bottom
  128|  5.80M|        for (x = 0, mask = 1U << bx4; x < w4; x++, mask <<= 1) {
  ------------------
  |  Branch (128:39): [True: 4.70M, False: 1.09M]
  ------------------
  129|  4.70M|            const int sidx = mask >= 0x10000U;
  130|  4.70M|            const unsigned smask = mask >> (sidx << 4);
  131|  4.70M|            int ttx = txa[1][0][0][x];
  132|  4.70M|            int step = txa[1][1][0][x];
  133|  5.63M|            for (y = step; y < h4; y += step) {
  ------------------
  |  Branch (133:28): [True: 925k, False: 4.70M]
  ------------------
  134|   925k|                const int btx = txa[1][0][y][x];
  135|   925k|                masks[1][by4 + y][imin(ttx, btx)][sidx] |= smask;
  136|   925k|                ttx = btx;
  137|   925k|                step = txa[1][1][y][x];
  138|   925k|            }
  139|  4.70M|        }
  140|  1.09M|    }
  141|       |
  142|  50.1M|    for (y = 0; y < h4; y++)
  ------------------
  |  Branch (142:17): [True: 46.5M, False: 3.62M]
  ------------------
  143|  46.5M|        l[y] = txa[0][0][y][w4 - 1];
  144|  3.62M|    memcpy(a, txa[1][0][h4 - 1], w4);
  145|  3.62M|}
lf_mask.c:decomp_tx:
   44|  12.3M|{
   45|  12.3M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[from];
   46|  12.3M|    const int is_split = (from == (int) TX_4X4 || depth > 1) ? 0 :
  ------------------
  |  Branch (46:27): [True: 6.38M, False: 5.97M]
  |  Branch (46:51): [True: 153k, False: 5.82M]
  ------------------
   47|  12.3M|        (tx_masks[depth] >> (y_off * 4 + x_off)) & 1;
   48|       |
   49|  12.3M|    if (is_split) {
  ------------------
  |  Branch (49:9): [True: 209k, False: 12.1M]
  ------------------
   50|   209k|        const enum RectTxfmSize sub = t_dim->sub;
   51|   209k|        const int htw4 = t_dim->w >> 1, hth4 = t_dim->h >> 1;
   52|       |
   53|   209k|        decomp_tx(txa, sub, depth + 1, y_off * 2 + 0, x_off * 2 + 0, tx_masks);
   54|   209k|        if (t_dim->w >= t_dim->h)
  ------------------
  |  Branch (54:13): [True: 164k, False: 45.0k]
  ------------------
   55|   164k|            decomp_tx((uint8_t(*)[2][32][32]) &txa[0][0][0][htw4],
   56|   164k|                      sub, depth + 1, y_off * 2 + 0, x_off * 2 + 1, tx_masks);
   57|   209k|        if (t_dim->h >= t_dim->w) {
  ------------------
  |  Branch (57:13): [True: 154k, False: 54.7k]
  ------------------
   58|   154k|            decomp_tx((uint8_t(*)[2][32][32]) &txa[0][0][hth4][0],
   59|   154k|                      sub, depth + 1, y_off * 2 + 1, x_off * 2 + 0, tx_masks);
   60|   154k|            if (t_dim->w >= t_dim->h)
  ------------------
  |  Branch (60:17): [True: 109k, False: 44.9k]
  ------------------
   61|   109k|                decomp_tx((uint8_t(*)[2][32][32]) &txa[0][0][hth4][htw4],
   62|   109k|                          sub, depth + 1, y_off * 2 + 1, x_off * 2 + 1, tx_masks);
   63|   154k|        }
   64|  12.1M|    } else {
   65|  12.1M|        const int lw = imin(2, t_dim->lw), lh = imin(2, t_dim->lh);
   66|       |
   67|  12.1M|#define set_ctx(rep_macro) \
   68|  12.1M|        for (int y = 0; y < t_dim->h; y++) { \
   69|  12.1M|            rep_macro(txa[0][0][y], 0, lw); \
   70|  12.1M|            rep_macro(txa[1][0][y], 0, lh); \
   71|  12.1M|            txa[0][1][y][0] = t_dim->w; \
   72|  12.1M|        }
   73|  12.1M|        case_set_upto16(t_dim->lw);
  ------------------
  |  |   80|  12.1M|    switch (var) { \
  |  |   81|  6.57M|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  |   68|  13.4M|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 6.91M, False: 6.57M]
  |  |  |  |  ------------------
  |  |  |  |   69|  6.91M|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|  6.91M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  6.91M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|  6.91M|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|  6.91M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  6.91M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|  6.91M|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|  6.91M|        }
  |  |  ------------------
  |  |  |  Branch (81:5): [True: 6.57M, False: 5.57M]
  |  |  ------------------
  |  |   82|   783k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  |   68|  2.79M|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 2.00M, False: 783k]
  |  |  |  |  ------------------
  |  |  |  |   69|  2.00M|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|  2.00M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  2.00M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|  2.00M|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|  2.00M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  2.00M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|  2.00M|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|  2.00M|        }
  |  |  ------------------
  |  |  |  Branch (82:5): [True: 783k, False: 11.3M]
  |  |  ------------------
  |  |   83|   707k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  |   68|  3.58M|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 2.87M, False: 707k]
  |  |  |  |  ------------------
  |  |  |  |   69|  2.87M|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|  2.87M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  2.87M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|  2.87M|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|  2.87M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  2.87M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|  2.87M|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|  2.87M|        }
  |  |  ------------------
  |  |  |  Branch (83:5): [True: 707k, False: 11.4M]
  |  |  ------------------
  |  |   84|   227k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  |   68|  1.60M|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 1.38M, False: 227k]
  |  |  |  |  ------------------
  |  |  |  |   69|  1.38M|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  1.38M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.38M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|  1.38M|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  1.38M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.38M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|  1.38M|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|  1.38M|        }
  |  |  ------------------
  |  |  |  Branch (84:5): [True: 227k, False: 11.9M]
  |  |  ------------------
  |  |   85|  3.86M|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  |   68|  65.3M|        for (int y = 0; y < t_dim->h; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (68:25): [True: 61.4M, False: 3.86M]
  |  |  |  |  ------------------
  |  |  |  |   69|  61.4M|            rep_macro(txa[0][0][y], 0, lw); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  61.4M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  61.4M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  61.4M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  61.4M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 61.4M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   70|  61.4M|            rep_macro(txa[1][0][y], 0, lh); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  61.4M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  61.4M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  61.4M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  61.4M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 61.4M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   71|  61.4M|            txa[0][1][y][0] = t_dim->w; \
  |  |  |  |   72|  61.4M|        }
  |  |  ------------------
  |  |  |  Branch (85:5): [True: 3.86M, False: 8.28M]
  |  |  ------------------
  |  |   86|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (86:5): [True: 0, False: 12.1M]
  |  |  ------------------
  |  |   87|  12.1M|    }
  ------------------
  |  Branch (73:9): [Folded, False: 0]
  ------------------
   74|  12.1M|#undef set_ctx
   75|  12.1M|        dav1d_memset_pow2[t_dim->lw](txa[1][1][0], t_dim->h);
   76|  12.1M|    }
   77|  12.3M|}
lf_mask.c:calc_lf_value:
  408|   658k|{
  409|   658k|    const int base = iclip(iclip(base_lvl + lf_delta, 0, 63) + seg_delta, 0, 63);
  410|       |
  411|   658k|    if (!mr_delta) {
  ------------------
  |  Branch (411:9): [True: 258k, False: 400k]
  ------------------
  412|   258k|        memset(lflvl_values, base, sizeof(*lflvl_values) * 8);
  413|   400k|    } else {
  414|   400k|        const int sh = base >= 32;
  415|   400k|        lflvl_values[0][0] = lflvl_values[0][1] =
  416|   400k|            iclip(base + (mr_delta->ref_delta[0] * (1 << sh)), 0, 63);
  417|  3.19M|        for (int r = 1; r < 8; r++) {
  ------------------
  |  Branch (417:25): [True: 2.79M, False: 400k]
  ------------------
  418|  8.38M|            for (int m = 0; m < 2; m++) {
  ------------------
  |  Branch (418:29): [True: 5.58M, False: 2.79M]
  ------------------
  419|  5.58M|                const int delta =
  420|  5.58M|                    mr_delta->mode_delta[m] + mr_delta->ref_delta[r];
  421|  5.58M|                lflvl_values[r][m] = iclip(base + (delta * (1 << sh)), 0, 63);
  422|  5.58M|            }
  423|  2.79M|        }
  424|   400k|    }
  425|   658k|}
lf_mask.c:calc_lf_value_chroma:
  431|   501k|{
  432|   501k|    if (!base_lvl)
  ------------------
  |  Branch (432:9): [True: 343k, False: 157k]
  ------------------
  433|   343k|        memset(lflvl_values, 0, sizeof(*lflvl_values) * 8);
  434|   157k|    else
  435|   157k|        calc_lf_value(lflvl_values, base_lvl, lf_delta, seg_delta, mr_delta);
  436|   501k|}

dav1d_version:
   61|  9.42k|COLD const char *dav1d_version(void) {
   62|  9.42k|    return DAV1D_VERSION;
  ------------------
  |  |    2|  9.42k|#define DAV1D_VERSION "1.5.4-0-g54706fc"
  ------------------
   63|  9.42k|}
dav1d_default_settings:
   71|  9.41k|COLD void dav1d_default_settings(Dav1dSettings *const s) {
   72|  9.41k|    s->n_threads = 0;
   73|  9.41k|    s->max_frame_delay = 0;
   74|  9.41k|    s->apply_grain = 1;
   75|  9.41k|    s->allocator.cookie = NULL;
   76|  9.41k|    s->allocator.alloc_picture_callback = dav1d_default_picture_alloc;
   77|  9.41k|    s->allocator.release_picture_callback = dav1d_default_picture_release;
   78|  9.41k|    s->logger.cookie = NULL;
   79|       |    s->logger.callback = dav1d_log_default_callback;
  ------------------
  |  |   43|  9.41k|#define dav1d_log_default_callback NULL
  ------------------
   80|  9.41k|    s->operating_point = 0;
   81|  9.41k|    s->all_layers = 1; // just until the tests are adjusted
   82|  9.41k|    s->frame_size_limit = 0;
   83|  9.41k|    s->strict_std_compliance = 0;
   84|  9.41k|    s->output_invisible_frames = 0;
   85|  9.41k|    s->inloop_filters = DAV1D_INLOOPFILTER_ALL;
   86|  9.41k|    s->decode_frame_type = DAV1D_DECODEFRAMETYPE_ALL;
   87|  9.41k|}
dav1d_open:
  140|  9.41k|COLD int dav1d_open(Dav1dContext **const c_out, const Dav1dSettings *const s) {
  141|  9.41k|    static pthread_once_t initted = PTHREAD_ONCE_INIT;
  142|  9.41k|    pthread_once(&initted, init_internal);
  143|       |
  144|  9.41k|    validate_input_or_ret(c_out != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  9.41k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 9.41k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  145|  9.41k|    validate_input_or_ret(s != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  9.41k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 9.41k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  146|  9.41k|    validate_input_or_ret(s->n_threads >= 0 &&
  ------------------
  |  |   52|  18.8k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 9.41k, False: 0]
  |  |  |  Branch (52:11): [True: 9.41k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  147|  9.41k|                          s->n_threads <= DAV1D_MAX_THREADS, DAV1D_ERR(EINVAL));
  148|  9.41k|    validate_input_or_ret(s->max_frame_delay >= 0 &&
  ------------------
  |  |   52|  18.8k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 9.41k, False: 0]
  |  |  |  Branch (52:11): [True: 9.41k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  149|  9.41k|                          s->max_frame_delay <= DAV1D_MAX_FRAME_DELAY, DAV1D_ERR(EINVAL));
  150|  9.41k|    validate_input_or_ret(s->allocator.alloc_picture_callback != NULL,
  ------------------
  |  |   52|  9.41k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 9.41k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  151|  9.41k|                          DAV1D_ERR(EINVAL));
  152|  9.41k|    validate_input_or_ret(s->allocator.release_picture_callback != NULL,
  ------------------
  |  |   52|  9.41k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 9.41k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  153|  9.41k|                          DAV1D_ERR(EINVAL));
  154|  9.41k|    validate_input_or_ret(s->operating_point >= 0 &&
  ------------------
  |  |   52|  18.8k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 9.41k, False: 0]
  |  |  |  Branch (52:11): [True: 9.41k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  155|  9.41k|                          s->operating_point <= 31, DAV1D_ERR(EINVAL));
  156|  9.41k|    validate_input_or_ret(s->decode_frame_type >= DAV1D_DECODEFRAMETYPE_ALL &&
  ------------------
  |  |   52|  18.8k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 9.41k, False: 0]
  |  |  |  Branch (52:11): [True: 9.41k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  157|  9.41k|                          s->decode_frame_type <= DAV1D_DECODEFRAMETYPE_KEY, DAV1D_ERR(EINVAL));
  158|       |
  159|  9.41k|    pthread_attr_t thread_attr;
  160|  9.41k|    if (pthread_attr_init(&thread_attr)) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (160:9): [True: 0, False: 9.41k]
  ------------------
  161|  9.41k|    size_t stack_size = 1024 * 1024 + get_stack_size_internal(&thread_attr);
  162|       |
  163|  9.41k|    pthread_attr_setstacksize(&thread_attr, stack_size);
  164|       |
  165|  9.41k|    Dav1dContext *const c = *c_out = dav1d_alloc_aligned(ALLOC_COMMON_CTX, sizeof(*c), 64);
  ------------------
  |  |  134|  9.41k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  166|  9.41k|    if (!c) goto error;
  ------------------
  |  Branch (166:9): [True: 0, False: 9.41k]
  ------------------
  167|  9.41k|    memset(c, 0, sizeof(*c));
  168|       |
  169|  9.41k|    c->allocator = s->allocator;
  170|  9.41k|    c->logger = s->logger;
  171|  9.41k|    c->apply_grain = s->apply_grain;
  172|  9.41k|    c->operating_point = s->operating_point;
  173|  9.41k|    c->all_layers = s->all_layers;
  174|  9.41k|    c->frame_size_limit = s->frame_size_limit;
  175|  9.41k|    c->strict_std_compliance = s->strict_std_compliance;
  176|  9.41k|    c->output_invisible_frames = s->output_invisible_frames;
  177|  9.41k|    c->inloop_filters = s->inloop_filters;
  178|  9.41k|    c->decode_frame_type = s->decode_frame_type;
  179|       |
  180|  9.41k|    dav1d_data_props_set_defaults(&c->cached_error_props);
  181|       |
  182|  9.41k|    if (dav1d_mem_pool_init(ALLOC_OBU_HDR, &c->seq_hdr_pool) ||
  ------------------
  |  |  131|  18.8k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 9.41k]
  |  |  ------------------
  ------------------
  183|  9.41k|        dav1d_mem_pool_init(ALLOC_OBU_HDR, &c->frame_hdr_pool) ||
  ------------------
  |  |  131|  18.8k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 9.41k]
  |  |  ------------------
  ------------------
  184|  9.41k|        dav1d_mem_pool_init(ALLOC_SEGMAP, &c->segmap_pool) ||
  ------------------
  |  |  131|  18.8k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 9.41k]
  |  |  ------------------
  ------------------
  185|  9.41k|        dav1d_mem_pool_init(ALLOC_REFMVS, &c->refmvs_pool) ||
  ------------------
  |  |  131|  18.8k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 9.41k]
  |  |  ------------------
  ------------------
  186|  9.41k|        dav1d_mem_pool_init(ALLOC_PIC_CTX, &c->pic_ctx_pool) ||
  ------------------
  |  |  131|  18.8k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 9.41k]
  |  |  ------------------
  ------------------
  187|  9.41k|        dav1d_mem_pool_init(ALLOC_CDF, &c->cdf_pool))
  ------------------
  |  |  131|  9.41k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 9.41k]
  |  |  ------------------
  ------------------
  188|      0|    {
  189|      0|        goto error;
  190|      0|    }
  191|       |
  192|  9.41k|    if (c->allocator.alloc_picture_callback   == dav1d_default_picture_alloc &&
  ------------------
  |  Branch (192:9): [True: 9.41k, False: 0]
  ------------------
  193|  9.41k|        c->allocator.release_picture_callback == dav1d_default_picture_release)
  ------------------
  |  Branch (193:9): [True: 9.41k, False: 0]
  ------------------
  194|  9.41k|    {
  195|  9.41k|        if (c->allocator.cookie) goto error;
  ------------------
  |  Branch (195:13): [True: 0, False: 9.41k]
  ------------------
  196|  9.41k|        if (dav1d_mem_pool_init(ALLOC_PIC, &c->picture_pool)) goto error;
  ------------------
  |  |  131|  9.41k|#define dav1d_mem_pool_init(type, pool) dav1d_mem_pool_init(pool)
  |  |  ------------------
  |  |  |  Branch (131:41): [True: 0, False: 9.41k]
  |  |  ------------------
  ------------------
  197|  9.41k|        c->allocator.cookie = c->picture_pool;
  198|  9.41k|    } else if (c->allocator.alloc_picture_callback   == dav1d_default_picture_alloc ||
  ------------------
  |  Branch (198:16): [True: 0, False: 0]
  ------------------
  199|      0|               c->allocator.release_picture_callback == dav1d_default_picture_release)
  ------------------
  |  Branch (199:16): [True: 0, False: 0]
  ------------------
  200|      0|    {
  201|      0|        goto error;
  202|      0|    }
  203|       |
  204|       |    /* On 32-bit systems extremely large frame sizes can cause overflows in
  205|       |     * dav1d_decode_frame() malloc size calculations. Prevent that from occuring
  206|       |     * by enforcing a maximum frame size limit, chosen to roughly correspond to
  207|       |     * the largest size possible to decode without exhausting virtual memory. */
  208|  9.41k|    if (sizeof(size_t) < 8 && s->frame_size_limit - 1 >= 8192 * 8192) {
  ------------------
  |  Branch (208:9): [Folded, False: 9.41k]
  |  Branch (208:31): [True: 0, False: 0]
  ------------------
  209|      0|        c->frame_size_limit = 8192 * 8192;
  210|      0|        if (s->frame_size_limit)
  ------------------
  |  Branch (210:13): [True: 0, False: 0]
  ------------------
  211|      0|            dav1d_log(c, "Frame size limit reduced from %u to %u.\n",
  ------------------
  |  |   44|      0|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 0]
  |  |  ------------------
  ------------------
  212|      0|                      s->frame_size_limit, c->frame_size_limit);
  213|      0|    }
  214|       |
  215|  9.41k|    c->flush = &c->flush_mem;
  216|  9.41k|    atomic_init(c->flush, 0);
  217|       |
  218|  9.41k|    get_num_threads(c, s, &c->n_tc, &c->n_fc);
  219|       |
  220|  9.41k|    c->fc = dav1d_alloc_aligned(ALLOC_THREAD_CTX, sizeof(*c->fc) * c->n_fc, 32);
  ------------------
  |  |  134|  9.41k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  221|  9.41k|    if (!c->fc) goto error;
  ------------------
  |  Branch (221:9): [True: 0, False: 9.41k]
  ------------------
  222|  9.41k|    memset(c->fc, 0, sizeof(*c->fc) * c->n_fc);
  223|       |
  224|  9.41k|    c->tc = dav1d_alloc_aligned(ALLOC_THREAD_CTX, sizeof(*c->tc) * c->n_tc, 64);
  ------------------
  |  |  134|  9.41k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  225|  9.41k|    if (!c->tc) goto error;
  ------------------
  |  Branch (225:9): [True: 0, False: 9.41k]
  ------------------
  226|  9.41k|    memset(c->tc, 0, sizeof(*c->tc) * c->n_tc);
  227|  9.41k|    if (c->n_tc > 1) {
  ------------------
  |  Branch (227:9): [True: 9.41k, False: 0]
  ------------------
  228|  9.41k|        if (pthread_mutex_init(&c->task_thread.lock, NULL)) goto error;
  ------------------
  |  Branch (228:13): [True: 0, False: 9.41k]
  ------------------
  229|  9.41k|        if (pthread_cond_init(&c->task_thread.cond, NULL)) {
  ------------------
  |  Branch (229:13): [True: 0, False: 9.41k]
  ------------------
  230|      0|            pthread_mutex_destroy(&c->task_thread.lock);
  231|      0|            goto error;
  232|      0|        }
  233|  9.41k|        if (pthread_cond_init(&c->task_thread.delayed_fg.cond, NULL)) {
  ------------------
  |  Branch (233:13): [True: 0, False: 9.41k]
  ------------------
  234|      0|            pthread_cond_destroy(&c->task_thread.cond);
  235|      0|            pthread_mutex_destroy(&c->task_thread.lock);
  236|      0|            goto error;
  237|      0|        }
  238|  9.41k|        c->task_thread.cur = c->n_fc;
  239|  9.41k|        atomic_init(&c->task_thread.reset_task_cur, UINT_MAX);
  240|  9.41k|        atomic_init(&c->task_thread.cond_signaled, 0);
  241|  9.41k|        c->task_thread.inited = 1;
  242|  9.41k|    }
  243|       |
  244|  9.41k|    if (c->n_fc > 1) {
  ------------------
  |  Branch (244:9): [True: 9.41k, False: 0]
  ------------------
  245|  9.41k|        const size_t out_delayed_sz = sizeof(*c->frame_thread.out_delayed) * c->n_fc;
  246|  9.41k|        c->frame_thread.out_delayed =
  247|  9.41k|            dav1d_malloc(ALLOC_THREAD_CTX, out_delayed_sz);
  ------------------
  |  |  132|  9.41k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
  248|  9.41k|        if (!c->frame_thread.out_delayed) goto error;
  ------------------
  |  Branch (248:13): [True: 0, False: 9.41k]
  ------------------
  249|  9.41k|        memset(c->frame_thread.out_delayed, 0, out_delayed_sz);
  250|  9.41k|    }
  251|  47.0k|    for (unsigned n = 0; n < c->n_fc; n++) {
  ------------------
  |  Branch (251:26): [True: 37.6k, False: 9.41k]
  ------------------
  252|  37.6k|        Dav1dFrameContext *const f = &c->fc[n];
  253|  37.6k|        if (c->n_tc > 1) {
  ------------------
  |  Branch (253:13): [True: 37.6k, False: 0]
  ------------------
  254|  37.6k|            if (pthread_mutex_init(&f->task_thread.lock, NULL)) goto error;
  ------------------
  |  Branch (254:17): [True: 0, False: 37.6k]
  ------------------
  255|  37.6k|            if (pthread_cond_init(&f->task_thread.cond, NULL)) {
  ------------------
  |  Branch (255:17): [True: 0, False: 37.6k]
  ------------------
  256|      0|                pthread_mutex_destroy(&f->task_thread.lock);
  257|      0|                goto error;
  258|      0|            }
  259|  37.6k|            if (pthread_mutex_init(&f->task_thread.pending_tasks.lock, NULL)) {
  ------------------
  |  Branch (259:17): [True: 0, False: 37.6k]
  ------------------
  260|      0|                pthread_cond_destroy(&f->task_thread.cond);
  261|      0|                pthread_mutex_destroy(&f->task_thread.lock);
  262|      0|                goto error;
  263|      0|            }
  264|  37.6k|        }
  265|  37.6k|        f->c = c;
  266|  37.6k|        f->task_thread.ttd = &c->task_thread;
  267|  37.6k|        f->lf.last_sharpness = -1;
  268|  37.6k|    }
  269|       |
  270|  47.0k|    for (unsigned m = 0; m < c->n_tc; m++) {
  ------------------
  |  Branch (270:26): [True: 37.6k, False: 9.41k]
  ------------------
  271|  37.6k|        Dav1dTaskContext *const t = &c->tc[m];
  272|  37.6k|        t->f = &c->fc[0];
  273|  37.6k|        t->task_thread.ttd = &c->task_thread;
  274|  37.6k|        t->c = c;
  275|  37.6k|        memset(t->cf_16bpc, 0, sizeof(t->cf_16bpc));
  276|  37.6k|        if (c->n_tc > 1) {
  ------------------
  |  Branch (276:13): [True: 37.6k, False: 0]
  ------------------
  277|  37.6k|            if (pthread_mutex_init(&t->task_thread.td.lock, NULL)) goto error;
  ------------------
  |  Branch (277:17): [True: 0, False: 37.6k]
  ------------------
  278|  37.6k|            if (pthread_cond_init(&t->task_thread.td.cond, NULL)) {
  ------------------
  |  Branch (278:17): [True: 0, False: 37.6k]
  ------------------
  279|      0|                pthread_mutex_destroy(&t->task_thread.td.lock);
  280|      0|                goto error;
  281|      0|            }
  282|  37.6k|            if (pthread_create(&t->task_thread.td.thread, &thread_attr, dav1d_worker_task, t)) {
  ------------------
  |  Branch (282:17): [True: 0, False: 37.6k]
  ------------------
  283|      0|                pthread_cond_destroy(&t->task_thread.td.cond);
  284|      0|                pthread_mutex_destroy(&t->task_thread.td.lock);
  285|      0|                goto error;
  286|      0|            }
  287|  37.6k|            t->task_thread.td.inited = 1;
  288|  37.6k|        }
  289|  37.6k|    }
  290|  9.41k|    dav1d_pal_dsp_init(&c->pal_dsp);
  291|  9.41k|    dav1d_refmvs_dsp_init(&c->refmvs_dsp);
  292|       |
  293|  9.41k|    pthread_attr_destroy(&thread_attr);
  294|       |
  295|  9.41k|    return 0;
  296|       |
  297|      0|error:
  298|      0|    if (c) close_internal(c_out, 0);
  ------------------
  |  Branch (298:9): [True: 0, False: 0]
  ------------------
  299|      0|    pthread_attr_destroy(&thread_attr);
  300|      0|    return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  301|  9.41k|}
dav1d_send_data:
  439|   430k|{
  440|   430k|    validate_input_or_ret(c != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|   430k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 430k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  441|   430k|    validate_input_or_ret(in != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|   430k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 430k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  442|       |
  443|   430k|    if (in->data) {
  ------------------
  |  Branch (443:9): [True: 430k, False: 0]
  ------------------
  444|   430k|        validate_input_or_ret(in->sz > 0 && in->sz <= SIZE_MAX / 2, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|   860k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 430k, False: 0]
  |  |  |  Branch (52:11): [True: 430k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  445|   430k|        c->drain = 0;
  446|   430k|    }
  447|   430k|    if (c->in.data)
  ------------------
  |  Branch (447:9): [True: 19.9k, False: 410k]
  ------------------
  448|  19.9k|        return DAV1D_ERR(EAGAIN);
  ------------------
  |  |   58|  19.9k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  449|   410k|    dav1d_data_ref(&c->in, in);
  450|       |
  451|   410k|    int res = gen_picture(c);
  452|   410k|    if (!res)
  ------------------
  |  Branch (452:9): [True: 370k, False: 39.8k]
  ------------------
  453|   370k|        dav1d_data_unref_internal(in);
  454|       |
  455|   410k|    return res;
  456|   430k|}
dav1d_get_picture:
  459|   411k|{
  460|   411k|    validate_input_or_ret(c != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|   411k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 411k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  461|   411k|    validate_input_or_ret(out != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|   411k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 411k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  462|       |
  463|   411k|    const int drain = c->drain;
  464|   411k|    c->drain = 1;
  465|       |
  466|   411k|    int res = gen_picture(c);
  467|   411k|    if (res < 0)
  ------------------
  |  Branch (467:9): [True: 141, False: 411k]
  ------------------
  468|    141|        return res;
  469|       |
  470|   411k|    if (c->cached_error) {
  ------------------
  |  Branch (470:9): [True: 176k, False: 234k]
  ------------------
  471|   176k|        const int res = c->cached_error;
  472|   176k|        c->cached_error = 0;
  473|   176k|        return res;
  474|   176k|    }
  475|       |
  476|   234k|    if (output_picture_ready(c, c->n_fc == 1))
  ------------------
  |  Branch (476:9): [True: 168k, False: 65.5k]
  ------------------
  477|   168k|        return output_image(c, out);
  478|       |
  479|  65.5k|    if (c->n_fc > 1 && drain)
  ------------------
  |  Branch (479:9): [True: 65.5k, False: 0]
  |  Branch (479:24): [True: 17.3k, False: 48.1k]
  ------------------
  480|  17.3k|        return drain_picture(c, out);
  481|       |
  482|  48.1k|    return DAV1D_ERR(EAGAIN);
  ------------------
  |  |   58|  48.1k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  483|  65.5k|}
dav1d_apply_grain:
  487|  10.1k|{
  488|  10.1k|    validate_input_or_ret(c != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  10.1k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 10.1k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  489|  10.1k|    validate_input_or_ret(out != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  10.1k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 10.1k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  490|  10.1k|    validate_input_or_ret(in != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  10.1k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 10.1k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  491|       |
  492|  10.1k|    if (!has_grain(in)) {
  ------------------
  |  Branch (492:9): [True: 0, False: 10.1k]
  ------------------
  493|      0|        dav1d_picture_ref(out, in);
  494|      0|        return 0;
  495|      0|    }
  496|       |
  497|  10.1k|    int res = dav1d_picture_alloc_copy(c, out, in->p.w, in);
  498|  10.1k|    if (res < 0) goto error;
  ------------------
  |  Branch (498:9): [True: 0, False: 10.1k]
  ------------------
  499|       |
  500|  10.1k|    if (c->n_tc > 1) {
  ------------------
  |  Branch (500:9): [True: 10.1k, False: 0]
  ------------------
  501|  10.1k|        dav1d_task_delayed_fg(c, out, in);
  502|  10.1k|    } else {
  503|      0|        switch (out->p.bpc) {
  504|      0|#if CONFIG_8BPC
  505|      0|        case 8:
  ------------------
  |  Branch (505:9): [True: 0, False: 0]
  ------------------
  506|      0|            dav1d_apply_grain_8bpc(&c->dsp[0].fg, out, in);
  507|      0|            break;
  508|      0|#endif
  509|      0|#if CONFIG_16BPC
  510|      0|        case 10:
  ------------------
  |  Branch (510:9): [True: 0, False: 0]
  ------------------
  511|      0|        case 12:
  ------------------
  |  Branch (511:9): [True: 0, False: 0]
  ------------------
  512|      0|            dav1d_apply_grain_16bpc(&c->dsp[(out->p.bpc >> 1) - 4].fg, out, in);
  513|      0|            break;
  514|      0|#endif
  515|      0|        default: abort();
  ------------------
  |  Branch (515:9): [True: 0, False: 0]
  ------------------
  516|      0|        }
  517|      0|    }
  518|       |
  519|  10.1k|    return 0;
  520|       |
  521|      0|error:
  522|      0|    dav1d_picture_unref_internal(out);
  523|      0|    return res;
  524|  10.1k|}
dav1d_flush:
  526|  9.41k|void dav1d_flush(Dav1dContext *const c) {
  527|  9.41k|    dav1d_data_unref_internal(&c->in);
  528|  9.41k|    if (c->out.p.frame_hdr)
  ------------------
  |  Branch (528:9): [True: 0, False: 9.41k]
  ------------------
  529|      0|        dav1d_thread_picture_unref(&c->out);
  530|  9.41k|    if (c->cache.p.frame_hdr)
  ------------------
  |  Branch (530:9): [True: 0, False: 9.41k]
  ------------------
  531|      0|        dav1d_thread_picture_unref(&c->cache);
  532|       |
  533|  9.41k|    c->drain = 0;
  534|  9.41k|    c->cached_error = 0;
  535|       |
  536|  84.7k|    for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (536:21): [True: 75.3k, False: 9.41k]
  ------------------
  537|  75.3k|        if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (537:13): [True: 65.8k, False: 9.48k]
  ------------------
  538|  65.8k|            dav1d_thread_picture_unref(&c->refs[i].p);
  539|  75.3k|        dav1d_ref_dec(&c->refs[i].segmap);
  540|  75.3k|        dav1d_ref_dec(&c->refs[i].refmvs);
  541|  75.3k|        dav1d_cdf_thread_unref(&c->cdf[i]);
  542|  75.3k|    }
  543|  9.41k|    c->frame_hdr = NULL;
  544|  9.41k|    c->seq_hdr = NULL;
  545|  9.41k|    dav1d_ref_dec(&c->seq_hdr_ref);
  546|       |
  547|  9.41k|    c->mastering_display = NULL;
  548|  9.41k|    c->content_light = NULL;
  549|  9.41k|    c->itut_t35 = NULL;
  550|  9.41k|    c->n_itut_t35 = 0;
  551|  9.41k|    dav1d_ref_dec(&c->mastering_display_ref);
  552|  9.41k|    dav1d_ref_dec(&c->content_light_ref);
  553|  9.41k|    dav1d_ref_dec(&c->itut_t35_ref);
  554|       |
  555|  9.41k|    dav1d_data_props_unref_internal(&c->cached_error_props);
  556|       |
  557|  9.41k|    if (c->n_fc == 1 && c->n_tc == 1) return;
  ------------------
  |  Branch (557:9): [True: 0, False: 9.41k]
  |  Branch (557:25): [True: 0, False: 0]
  ------------------
  558|  9.41k|    atomic_store(c->flush, 1);
  559|       |
  560|  9.41k|    if (c->n_tc > 1) {
  ------------------
  |  Branch (560:9): [True: 9.41k, False: 0]
  ------------------
  561|  9.41k|        pthread_mutex_lock(&c->task_thread.lock);
  562|       |        // stop running tasks in worker threads
  563|  47.0k|        for (unsigned i = 0; i < c->n_tc; i++) {
  ------------------
  |  Branch (563:30): [True: 37.6k, False: 9.41k]
  ------------------
  564|  37.6k|            Dav1dTaskContext *const tc = &c->tc[i];
  565|  40.0k|            while (!tc->task_thread.flushed) {
  ------------------
  |  Branch (565:20): [True: 2.33k, False: 37.6k]
  ------------------
  566|  2.33k|                pthread_cond_wait(&tc->task_thread.td.cond, &c->task_thread.lock);
  567|  2.33k|            }
  568|  37.6k|        }
  569|  47.0k|        for (unsigned i = 0; i < c->n_fc; i++) {
  ------------------
  |  Branch (569:30): [True: 37.6k, False: 9.41k]
  ------------------
  570|  37.6k|            c->fc[i].task_thread.task_head = NULL;
  571|  37.6k|            c->fc[i].task_thread.task_tail = NULL;
  572|  37.6k|            c->fc[i].task_thread.task_cur_prev = NULL;
  573|  37.6k|            c->fc[i].task_thread.pending_tasks.head = NULL;
  574|  37.6k|            c->fc[i].task_thread.pending_tasks.tail = NULL;
  575|  37.6k|            atomic_init(&c->fc[i].task_thread.pending_tasks.merge, 0);
  576|  37.6k|        }
  577|  9.41k|        atomic_init(&c->task_thread.first, 0);
  578|  9.41k|        c->task_thread.cur = c->n_fc;
  579|  9.41k|        atomic_store(&c->task_thread.reset_task_cur, UINT_MAX);
  580|  9.41k|        atomic_store(&c->task_thread.cond_signaled, 0);
  581|  9.41k|        pthread_mutex_unlock(&c->task_thread.lock);
  582|  9.41k|    }
  583|       |
  584|  9.41k|    if (c->n_fc > 1) {
  ------------------
  |  Branch (584:9): [True: 9.41k, False: 0]
  ------------------
  585|  47.0k|        for (unsigned n = 0, next = c->frame_thread.next; n < c->n_fc; n++, next++) {
  ------------------
  |  Branch (585:59): [True: 37.6k, False: 9.41k]
  ------------------
  586|  37.6k|            if (next == c->n_fc) next = 0;
  ------------------
  |  Branch (586:17): [True: 7.81k, False: 29.8k]
  ------------------
  587|  37.6k|            Dav1dFrameContext *const f = &c->fc[next];
  588|  37.6k|            dav1d_decode_frame_exit(f, -1);
  589|  37.6k|            f->n_tile_data = 0;
  590|  37.6k|            f->task_thread.retval = 0;
  591|  37.6k|            f->task_thread.error = 0;
  592|  37.6k|            Dav1dThreadPicture *out_delayed = &c->frame_thread.out_delayed[next];
  593|  37.6k|            if (out_delayed->p.frame_hdr) {
  ------------------
  |  Branch (593:17): [True: 4.31k, False: 33.3k]
  ------------------
  594|  4.31k|                dav1d_thread_picture_unref(out_delayed);
  595|  4.31k|            }
  596|  37.6k|        }
  597|  9.41k|        c->frame_thread.next = 0;
  598|  9.41k|    }
  599|       |    atomic_store(c->flush, 0);
  600|  9.41k|}
dav1d_close:
  602|  9.41k|COLD void dav1d_close(Dav1dContext **const c_out) {
  603|  9.41k|    validate_input(c_out != NULL);
  ------------------
  |  |   59|  9.41k|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|  9.41k|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 9.41k]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  604|       |#if TRACK_HEAP_ALLOCATIONS
  605|       |    dav1d_log_alloc_stats(*c_out);
  606|       |#endif
  607|  9.41k|    close_internal(c_out, 1);
  608|  9.41k|}
dav1d_picture_unref:
  727|   176k|void dav1d_picture_unref(Dav1dPicture *const p) {
  728|   176k|    dav1d_picture_unref_internal(p);
  729|   176k|}
dav1d_data_create:
  731|   411k|uint8_t *dav1d_data_create(Dav1dData *const buf, const size_t sz) {
  732|   411k|    return dav1d_data_create_internal(buf, sz);
  733|   411k|}
dav1d_data_unref:
  756|  41.2k|void dav1d_data_unref(Dav1dData *const buf) {
  757|  41.2k|    dav1d_data_unref_internal(buf);
  758|  41.2k|}
lib.c:get_num_threads:
  111|  9.41k|{
  112|       |    /* ceil(sqrt(n)) */
  113|  9.41k|    static const uint8_t fc_lut[49] = {
  114|  9.41k|        1,                                     /*     1 */
  115|  9.41k|        2, 2, 2,                               /*  2- 4 */
  116|  9.41k|        3, 3, 3, 3, 3,                         /*  5- 9 */
  117|  9.41k|        4, 4, 4, 4, 4, 4, 4,                   /* 10-16 */
  118|  9.41k|        5, 5, 5, 5, 5, 5, 5, 5, 5,             /* 17-25 */
  119|  9.41k|        6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,       /* 26-36 */
  120|  9.41k|        7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, /* 37-49 */
  121|  9.41k|    };
  122|  9.41k|    *n_tc = s->n_threads ? s->n_threads :
  ------------------
  |  Branch (122:13): [True: 9.41k, False: 0]
  ------------------
  123|  9.41k|        iclip(dav1d_num_logical_processors(c), 1, DAV1D_MAX_THREADS);
  ------------------
  |  |   46|      0|#define DAV1D_MAX_THREADS 256
  ------------------
  124|  9.41k|    *n_fc = s->max_frame_delay ? umin(s->max_frame_delay, *n_tc) :
  ------------------
  |  Branch (124:13): [True: 9.41k, False: 0]
  ------------------
  125|  9.41k|            *n_tc < 50 ? fc_lut[*n_tc - 1] : 8; // min(8, ceil(sqrt(n)))
  ------------------
  |  Branch (125:13): [True: 0, False: 0]
  ------------------
  126|  9.41k|}
lib.c:init_internal:
   53|      1|static COLD void init_internal(void) {
   54|      1|    dav1d_init_cpu();
   55|      1|    dav1d_init_ii_wedge_masks();
   56|      1|    dav1d_init_intra_edge_tree();
   57|      1|    dav1d_init_qm_tables();
   58|      1|    dav1d_init_thread();
  ------------------
  |  |  144|      1|#define dav1d_init_thread() do {} while (0)
  |  |  ------------------
  |  |  |  Branch (144:42): [Folded, False: 1]
  |  |  ------------------
  ------------------
   59|      1|}
lib.c:get_stack_size_internal:
   93|  9.41k|static COLD size_t get_stack_size_internal(const pthread_attr_t *const thread_attr) {
   94|       |    /* glibc has an issue where the size of the TLS is subtracted from the stack
   95|       |     * size instead of allocated separately. As a result the specified stack
   96|       |     * size may be insufficient when used in an application with large amounts
   97|       |     * of TLS data. The following is a workaround to compensate for that.
   98|       |     * See https://sourceware.org/bugzilla/show_bug.cgi?id=11787 */
   99|  9.41k|    size_t (*const get_minstack)(const pthread_attr_t*) =
  100|  9.41k|        dlsym(RTLD_DEFAULT, "__pthread_get_minstack");
  101|  9.41k|    if (get_minstack)
  ------------------
  |  Branch (101:9): [True: 9.41k, False: 0]
  ------------------
  102|  9.41k|        return get_minstack(thread_attr) - PTHREAD_STACK_MIN;
  103|      0|    return 0;
  104|  9.41k|}
lib.c:gen_picture:
  413|   821k|{
  414|   821k|    Dav1dData *const in = &c->in;
  415|       |
  416|   821k|    if (output_picture_ready(c, 0))
  ------------------
  |  Branch (416:9): [True: 337k, False: 484k]
  ------------------
  417|   337k|        return 0;
  418|       |
  419|   633k|    while (in->sz > 0) {
  ------------------
  |  Branch (419:12): [True: 534k, False: 98.8k]
  ------------------
  420|   534k|        const ptrdiff_t res = dav1d_parse_obus(c, in);
  421|   534k|        if (res < 0) {
  ------------------
  |  Branch (421:13): [True: 41.0k, False: 493k]
  ------------------
  422|  41.0k|            dav1d_data_unref_internal(in);
  423|   493k|        } else {
  424|   493k|            assert((size_t)res <= in->sz);
  ------------------
  |  Branch (424:13): [True: 493k, False: 0]
  ------------------
  425|   493k|            in->sz -= res;
  426|   493k|            in->data += res;
  427|   493k|            if (!in->sz) dav1d_data_unref_internal(in);
  ------------------
  |  Branch (427:17): [True: 369k, False: 124k]
  ------------------
  428|   493k|        }
  429|   534k|        if (output_picture_ready(c, 0))
  ------------------
  |  Branch (429:13): [True: 345k, False: 188k]
  ------------------
  430|   345k|            break;
  431|   188k|        if (res < 0)
  ------------------
  |  Branch (431:13): [True: 40.0k, False: 148k]
  ------------------
  432|  40.0k|            return (int)res;
  433|   188k|    }
  434|       |
  435|   444k|    return 0;
  436|   484k|}
lib.c:output_picture_ready:
  332|  1.60M|static int output_picture_ready(Dav1dContext *const c, const int drain) {
  333|  1.60M|    if (c->cached_error) return 1;
  ------------------
  |  Branch (333:9): [True: 351k, False: 1.25M]
  ------------------
  334|  1.25M|    if (!c->all_layers && c->max_spatial_id) {
  ------------------
  |  Branch (334:9): [True: 0, False: 1.25M]
  |  Branch (334:27): [True: 0, False: 0]
  ------------------
  335|      0|        if (c->out.p.data[0] && c->cache.p.data[0]) {
  ------------------
  |  Branch (335:13): [True: 0, False: 0]
  |  Branch (335:33): [True: 0, False: 0]
  ------------------
  336|      0|            if (c->max_spatial_id == c->cache.p.frame_hdr->spatial_id ||
  ------------------
  |  Branch (336:17): [True: 0, False: 0]
  ------------------
  337|      0|                c->out.flags & PICTURE_FLAG_NEW_TEMPORAL_UNIT)
  ------------------
  |  Branch (337:17): [True: 0, False: 0]
  ------------------
  338|      0|                return 1;
  339|      0|            dav1d_thread_picture_unref(&c->cache);
  340|      0|            dav1d_thread_picture_move_ref(&c->cache, &c->out);
  341|      0|            return 0;
  342|      0|        } else if (c->cache.p.data[0] && drain) {
  ------------------
  |  Branch (342:20): [True: 0, False: 0]
  |  Branch (342:42): [True: 0, False: 0]
  ------------------
  343|      0|            return 1;
  344|      0|        } else if (c->out.p.data[0]) {
  ------------------
  |  Branch (344:20): [True: 0, False: 0]
  ------------------
  345|      0|            dav1d_thread_picture_move_ref(&c->cache, &c->out);
  346|      0|            return 0;
  347|      0|        }
  348|      0|    }
  349|       |
  350|  1.25M|    return !!c->out.p.data[0];
  351|  1.25M|}
lib.c:output_image:
  312|   176k|{
  313|   176k|    int res = 0;
  314|       |
  315|   176k|    Dav1dThreadPicture *const in = (c->all_layers || !c->max_spatial_id)
  ------------------
  |  Branch (315:37): [True: 176k, False: 0]
  |  Branch (315:54): [True: 0, False: 0]
  ------------------
  316|   176k|                                   ? &c->out : &c->cache;
  317|   176k|    if (!c->apply_grain || !has_grain(&in->p)) {
  ------------------
  |  Branch (317:9): [True: 0, False: 176k]
  |  Branch (317:28): [True: 166k, False: 10.1k]
  ------------------
  318|   166k|        dav1d_picture_move_ref(out, &in->p);
  319|   166k|        dav1d_thread_picture_unref(in);
  320|   166k|        goto end;
  321|   166k|    }
  322|       |
  323|  10.1k|    res = dav1d_apply_grain(c, out, &in->p);
  324|  10.1k|    dav1d_thread_picture_unref(in);
  325|   176k|end:
  326|   176k|    if (!c->all_layers && c->max_spatial_id && c->out.p.data[0]) {
  ------------------
  |  Branch (326:9): [True: 0, False: 176k]
  |  Branch (326:27): [True: 0, False: 0]
  |  Branch (326:48): [True: 0, False: 0]
  ------------------
  327|      0|        dav1d_thread_picture_move_ref(in, &c->out);
  328|      0|    }
  329|   176k|    return res;
  330|  10.1k|}
lib.c:drain_picture:
  353|  17.3k|static int drain_picture(Dav1dContext *const c, Dav1dPicture *const out) {
  354|  17.3k|    unsigned drain_count = 0;
  355|  17.3k|    int drained = 0;
  356|  48.9k|    do {
  357|  48.9k|        const unsigned next = c->frame_thread.next;
  358|  48.9k|        Dav1dFrameContext *const f = &c->fc[next];
  359|  48.9k|        pthread_mutex_lock(&c->task_thread.lock);
  360|  59.0k|        while (f->n_tile_data > 0)
  ------------------
  |  Branch (360:16): [True: 10.0k, False: 48.9k]
  ------------------
  361|  10.0k|            pthread_cond_wait(&f->task_thread.cond,
  362|  10.0k|                              &f->task_thread.ttd->lock);
  363|  48.9k|        Dav1dThreadPicture *const out_delayed =
  364|  48.9k|            &c->frame_thread.out_delayed[next];
  365|  48.9k|        if (out_delayed->p.data[0] || atomic_load(&f->task_thread.error)) {
  ------------------
  |  Branch (365:13): [True: 13.0k, False: 35.9k]
  |  Branch (365:39): [True: 53, False: 35.8k]
  ------------------
  366|  13.0k|            unsigned first = atomic_load(&c->task_thread.first);
  367|  13.0k|            if (first + 1U < c->n_fc)
  ------------------
  |  Branch (367:17): [True: 12.0k, False: 1.03k]
  ------------------
  368|  13.0k|                atomic_fetch_add(&c->task_thread.first, 1U);
  369|  1.03k|            else
  370|  13.0k|                atomic_store(&c->task_thread.first, 0);
  371|  13.0k|            atomic_compare_exchange_strong(&c->task_thread.reset_task_cur,
  372|  13.0k|                                           &first, UINT_MAX);
  373|  13.0k|            if (c->task_thread.cur && c->task_thread.cur < c->n_fc)
  ------------------
  |  Branch (373:17): [True: 13.0k, False: 14]
  |  Branch (373:39): [True: 1.92k, False: 11.1k]
  ------------------
  374|  1.92k|                c->task_thread.cur--;
  375|  13.0k|            drained = 1;
  376|  35.8k|        } else if (drained) {
  ------------------
  |  Branch (376:20): [True: 350, False: 35.5k]
  ------------------
  377|    350|            pthread_mutex_unlock(&c->task_thread.lock);
  378|    350|            break;
  379|    350|        }
  380|  48.5k|        if (++c->frame_thread.next == c->n_fc)
  ------------------
  |  Branch (380:13): [True: 12.1k, False: 36.4k]
  ------------------
  381|  12.1k|            c->frame_thread.next = 0;
  382|  48.5k|        pthread_mutex_unlock(&c->task_thread.lock);
  383|  48.5k|        const int error = f->task_thread.retval;
  384|  48.5k|        if (error) {
  ------------------
  |  Branch (384:13): [True: 4.11k, False: 44.4k]
  ------------------
  385|  4.11k|            f->task_thread.retval = 0;
  386|  4.11k|            dav1d_data_props_copy(&c->cached_error_props, &out_delayed->p.m);
  387|  4.11k|            dav1d_thread_picture_unref(out_delayed);
  388|  4.11k|            return error;
  389|  4.11k|        }
  390|  44.4k|        if (out_delayed->p.data[0]) {
  ------------------
  |  Branch (390:13): [True: 8.92k, False: 35.5k]
  ------------------
  391|  8.92k|            const unsigned progress =
  392|  8.92k|                atomic_load_explicit(&out_delayed->progress[1],
  393|  8.92k|                                     memory_order_relaxed);
  394|  8.92k|            if ((out_delayed->visible || c->output_invisible_frames) &&
  ------------------
  |  Branch (394:18): [True: 7.83k, False: 1.08k]
  |  Branch (394:42): [True: 0, False: 1.08k]
  ------------------
  395|  7.83k|                progress != FRAME_ERROR)
  ------------------
  |  |   35|  7.83k|#define FRAME_ERROR (UINT_MAX - 1)
  ------------------
  |  Branch (395:17): [True: 7.68k, False: 151]
  ------------------
  396|  7.68k|            {
  397|  7.68k|                dav1d_thread_picture_ref(&c->out, out_delayed);
  398|  7.68k|                c->event_flags |= dav1d_picture_get_event_flags(out_delayed);
  399|  7.68k|            }
  400|  8.92k|            dav1d_thread_picture_unref(out_delayed);
  401|  8.92k|            if (output_picture_ready(c, 0))
  ------------------
  |  Branch (401:17): [True: 7.68k, False: 1.23k]
  ------------------
  402|  7.68k|                return output_image(c, out);
  403|  8.92k|        }
  404|  44.4k|    } while (++drain_count < c->n_fc);
  ------------------
  |  Branch (404:14): [True: 31.6k, False: 5.19k]
  ------------------
  405|       |
  406|  5.54k|    if (output_picture_ready(c, 1))
  ------------------
  |  Branch (406:9): [True: 0, False: 5.54k]
  ------------------
  407|      0|        return output_image(c, out);
  408|       |
  409|  5.54k|    return DAV1D_ERR(EAGAIN);
  ------------------
  |  |   58|  5.54k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  410|  5.54k|}
lib.c:has_grain:
  304|   186k|{
  305|   186k|    const Dav1dFilmGrainData *fgdata = &pic->frame_hdr->film_grain.data;
  306|   186k|    return fgdata->num_y_points || fgdata->num_uv_points[0] ||
  ------------------
  |  Branch (306:12): [True: 9.13k, False: 177k]
  |  Branch (306:36): [True: 1.96k, False: 175k]
  ------------------
  307|   175k|           fgdata->num_uv_points[1] || (fgdata->clip_to_restricted_range &&
  ------------------
  |  Branch (307:12): [True: 5.67k, False: 169k]
  |  Branch (307:41): [True: 4.05k, False: 165k]
  ------------------
  308|  4.05k|                                        fgdata->chroma_scaling_from_luma);
  ------------------
  |  Branch (308:41): [True: 3.45k, False: 593]
  ------------------
  309|   186k|}
lib.c:close_internal:
  610|  9.41k|static COLD void close_internal(Dav1dContext **const c_out, int flush) {
  611|  9.41k|    Dav1dContext *const c = *c_out;
  612|  9.41k|    if (!c) return;
  ------------------
  |  Branch (612:9): [True: 0, False: 9.41k]
  ------------------
  613|       |
  614|  9.41k|    if (flush) dav1d_flush(c);
  ------------------
  |  Branch (614:9): [True: 9.41k, False: 0]
  ------------------
  615|       |
  616|  9.41k|    if (c->tc) {
  ------------------
  |  Branch (616:9): [True: 9.41k, False: 0]
  ------------------
  617|  9.41k|        struct TaskThreadData *ttd = &c->task_thread;
  618|  9.41k|        if (ttd->inited) {
  ------------------
  |  Branch (618:13): [True: 9.41k, False: 0]
  ------------------
  619|  9.41k|            pthread_mutex_lock(&ttd->lock);
  620|  47.0k|            for (unsigned n = 0; n < c->n_tc && c->tc[n].task_thread.td.inited; n++)
  ------------------
  |  Branch (620:34): [True: 37.6k, False: 9.41k]
  |  Branch (620:49): [True: 37.6k, False: 0]
  ------------------
  621|  37.6k|                c->tc[n].task_thread.die = 1;
  622|  9.41k|            pthread_cond_broadcast(&ttd->cond);
  623|  9.41k|            pthread_mutex_unlock(&ttd->lock);
  624|  47.0k|            for (unsigned n = 0; n < c->n_tc; n++) {
  ------------------
  |  Branch (624:34): [True: 37.6k, False: 9.41k]
  ------------------
  625|  37.6k|                Dav1dTaskContext *const pf = &c->tc[n];
  626|  37.6k|                if (!pf->task_thread.td.inited) break;
  ------------------
  |  Branch (626:21): [True: 0, False: 37.6k]
  ------------------
  627|  37.6k|                pthread_join(pf->task_thread.td.thread, NULL);
  628|  37.6k|                pthread_cond_destroy(&pf->task_thread.td.cond);
  629|  37.6k|                pthread_mutex_destroy(&pf->task_thread.td.lock);
  630|  37.6k|            }
  631|  9.41k|            pthread_cond_destroy(&ttd->delayed_fg.cond);
  632|  9.41k|            pthread_cond_destroy(&ttd->cond);
  633|  9.41k|            pthread_mutex_destroy(&ttd->lock);
  634|  9.41k|        }
  635|  9.41k|        dav1d_free_aligned(c->tc);
  ------------------
  |  |  136|  9.41k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  636|  9.41k|    }
  637|       |
  638|  47.0k|    for (unsigned n = 0; c->fc && n < c->n_fc; n++) {
  ------------------
  |  Branch (638:26): [True: 47.0k, False: 0]
  |  Branch (638:35): [True: 37.6k, False: 9.41k]
  ------------------
  639|  37.6k|        Dav1dFrameContext *const f = &c->fc[n];
  640|       |
  641|       |        // clean-up threading stuff
  642|  37.6k|        if (c->n_fc > 1) {
  ------------------
  |  Branch (642:13): [True: 37.6k, False: 0]
  ------------------
  643|  37.6k|            dav1d_free(f->tile_thread.lowest_pixel_mem);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  644|  37.6k|            dav1d_free(f->frame_thread.b);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  645|  37.6k|            dav1d_free_aligned(f->frame_thread.cbi);
  ------------------
  |  |  136|  37.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  646|  37.6k|            dav1d_free_aligned(f->frame_thread.pal_idx);
  ------------------
  |  |  136|  37.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  647|  37.6k|            dav1d_free_aligned(f->frame_thread.cf);
  ------------------
  |  |  136|  37.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  648|  37.6k|            dav1d_free(f->frame_thread.tile_start_off);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  649|  37.6k|            dav1d_free_aligned(f->frame_thread.pal);
  ------------------
  |  |  136|  37.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  650|  37.6k|        }
  651|  37.6k|        if (c->n_tc > 1) {
  ------------------
  |  Branch (651:13): [True: 37.6k, False: 0]
  ------------------
  652|  37.6k|            pthread_mutex_destroy(&f->task_thread.pending_tasks.lock);
  653|  37.6k|            pthread_cond_destroy(&f->task_thread.cond);
  654|  37.6k|            pthread_mutex_destroy(&f->task_thread.lock);
  655|  37.6k|        }
  656|  37.6k|        dav1d_free(f->frame_thread.frame_progress);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  657|  37.6k|        dav1d_free(f->task_thread.tasks);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  658|  37.6k|        dav1d_free(f->task_thread.tile_tasks[0]);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  659|  37.6k|        dav1d_free_aligned(f->ts);
  ------------------
  |  |  136|  37.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  660|  37.6k|        dav1d_free_aligned(f->ipred_edge[0]);
  ------------------
  |  |  136|  37.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  661|  37.6k|        dav1d_free(f->a);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  662|  37.6k|        dav1d_free(f->tile);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  663|  37.6k|        dav1d_free(f->lf.mask);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  664|  37.6k|        dav1d_free(f->lf.level);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  665|  37.6k|        dav1d_free(f->lf.lr_mask);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  666|  37.6k|        dav1d_free(f->lf.tx_lpf_right_edge[0]);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  667|  37.6k|        dav1d_free(f->lf.start_of_tile_row);
  ------------------
  |  |  135|  37.6k|#define dav1d_free(ptr) free(ptr)
  ------------------
  668|  37.6k|        dav1d_free_aligned(f->rf.r);
  ------------------
  |  |  136|  37.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  669|  37.6k|        dav1d_free_aligned(f->lf.cdef_line_buf);
  ------------------
  |  |  136|  37.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  670|  37.6k|        dav1d_free_aligned(f->lf.lr_line_buf);
  ------------------
  |  |  136|  37.6k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  671|  37.6k|    }
  672|  9.41k|    dav1d_free_aligned(c->fc);
  ------------------
  |  |  136|  9.41k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  673|  9.41k|    if (c->n_fc > 1 && c->frame_thread.out_delayed) {
  ------------------
  |  Branch (673:9): [True: 9.41k, False: 0]
  |  Branch (673:24): [True: 9.41k, False: 0]
  ------------------
  674|  47.0k|        for (unsigned n = 0; n < c->n_fc; n++)
  ------------------
  |  Branch (674:30): [True: 37.6k, False: 9.41k]
  ------------------
  675|  37.6k|            if (c->frame_thread.out_delayed[n].p.frame_hdr)
  ------------------
  |  Branch (675:17): [True: 0, False: 37.6k]
  ------------------
  676|      0|                dav1d_thread_picture_unref(&c->frame_thread.out_delayed[n]);
  677|  9.41k|        dav1d_free(c->frame_thread.out_delayed);
  ------------------
  |  |  135|  9.41k|#define dav1d_free(ptr) free(ptr)
  ------------------
  678|  9.41k|    }
  679|  9.52k|    for (int n = 0; n < c->n_tile_data; n++)
  ------------------
  |  Branch (679:21): [True: 104, False: 9.41k]
  ------------------
  680|    104|        dav1d_data_unref_internal(&c->tile[n].data);
  681|  9.41k|    dav1d_free(c->tile);
  ------------------
  |  |  135|  9.41k|#define dav1d_free(ptr) free(ptr)
  ------------------
  682|  84.7k|    for (int n = 0; n < 8; n++) {
  ------------------
  |  Branch (682:21): [True: 75.3k, False: 9.41k]
  ------------------
  683|  75.3k|        dav1d_cdf_thread_unref(&c->cdf[n]);
  684|  75.3k|        if (c->refs[n].p.p.frame_hdr)
  ------------------
  |  Branch (684:13): [True: 0, False: 75.3k]
  ------------------
  685|      0|            dav1d_thread_picture_unref(&c->refs[n].p);
  686|  75.3k|        dav1d_ref_dec(&c->refs[n].refmvs);
  687|  75.3k|        dav1d_ref_dec(&c->refs[n].segmap);
  688|  75.3k|    }
  689|  9.41k|    dav1d_ref_dec(&c->seq_hdr_ref);
  690|  9.41k|    dav1d_ref_dec(&c->frame_hdr_ref);
  691|       |
  692|  9.41k|    dav1d_ref_dec(&c->mastering_display_ref);
  693|  9.41k|    dav1d_ref_dec(&c->content_light_ref);
  694|  9.41k|    dav1d_ref_dec(&c->itut_t35_ref);
  695|       |
  696|  9.41k|    dav1d_mem_pool_end(c->seq_hdr_pool);
  697|  9.41k|    dav1d_mem_pool_end(c->frame_hdr_pool);
  698|  9.41k|    dav1d_mem_pool_end(c->segmap_pool);
  699|  9.41k|    dav1d_mem_pool_end(c->refmvs_pool);
  700|  9.41k|    dav1d_mem_pool_end(c->cdf_pool);
  701|  9.41k|    dav1d_mem_pool_end(c->picture_pool);
  702|  9.41k|    dav1d_mem_pool_end(c->pic_ctx_pool);
  703|       |
  704|  9.41k|    dav1d_freep_aligned(c_out);
  705|  9.41k|}

dav1d_loop_filter_dsp_init_8bpc:
  259|  3.46k|COLD void bitfn(dav1d_loop_filter_dsp_init)(Dav1dLoopFilterDSPContext *const c) {
  260|  3.46k|    c->loop_filter_sb[0][0] = loop_filter_h_sb128y_c;
  261|  3.46k|    c->loop_filter_sb[0][1] = loop_filter_v_sb128y_c;
  262|  3.46k|    c->loop_filter_sb[1][0] = loop_filter_h_sb128uv_c;
  263|  3.46k|    c->loop_filter_sb[1][1] = loop_filter_v_sb128uv_c;
  264|       |
  265|  3.46k|#if HAVE_ASM
  266|       |#if ARCH_AARCH64 || ARCH_ARM
  267|       |    loop_filter_dsp_init_arm(c);
  268|       |#elif ARCH_LOONGARCH64
  269|       |    loop_filter_dsp_init_loongarch(c);
  270|       |#elif ARCH_PPC64LE
  271|       |    loop_filter_dsp_init_ppc(c);
  272|       |#elif ARCH_X86
  273|       |    loop_filter_dsp_init_x86(c);
  274|  3.46k|#endif
  275|  3.46k|#endif
  276|  3.46k|}
dav1d_loop_filter_dsp_init_16bpc:
  259|  5.11k|COLD void bitfn(dav1d_loop_filter_dsp_init)(Dav1dLoopFilterDSPContext *const c) {
  260|  5.11k|    c->loop_filter_sb[0][0] = loop_filter_h_sb128y_c;
  261|  5.11k|    c->loop_filter_sb[0][1] = loop_filter_v_sb128y_c;
  262|  5.11k|    c->loop_filter_sb[1][0] = loop_filter_h_sb128uv_c;
  263|  5.11k|    c->loop_filter_sb[1][1] = loop_filter_v_sb128uv_c;
  264|       |
  265|  5.11k|#if HAVE_ASM
  266|       |#if ARCH_AARCH64 || ARCH_ARM
  267|       |    loop_filter_dsp_init_arm(c);
  268|       |#elif ARCH_LOONGARCH64
  269|       |    loop_filter_dsp_init_loongarch(c);
  270|       |#elif ARCH_PPC64LE
  271|       |    loop_filter_dsp_init_ppc(c);
  272|       |#elif ARCH_X86
  273|       |    loop_filter_dsp_init_x86(c);
  274|  5.11k|#endif
  275|  5.11k|#endif
  276|  5.11k|}

dav1d_loop_restoration_dsp_init_8bpc:
 1367|  3.46k|{
 1368|  3.46k|    c->wiener[0] = c->wiener[1] = wiener_c;
 1369|  3.46k|    c->sgr[0] = sgr_5x5_c;
 1370|  3.46k|    c->sgr[1] = sgr_3x3_c;
 1371|  3.46k|    c->sgr[2] = sgr_mix_c;
 1372|       |
 1373|  3.46k|#if HAVE_ASM
 1374|       |#if ARCH_AARCH64 || ARCH_ARM
 1375|       |    loop_restoration_dsp_init_arm(c, bpc);
 1376|       |#elif ARCH_LOONGARCH64
 1377|       |    loop_restoration_dsp_init_loongarch(c, bpc);
 1378|       |#elif ARCH_PPC64LE
 1379|       |    loop_restoration_dsp_init_ppc(c, bpc);
 1380|       |#elif ARCH_X86
 1381|       |    loop_restoration_dsp_init_x86(c, bpc);
 1382|  3.46k|#endif
 1383|  3.46k|#endif
 1384|  3.46k|}
looprestoration_tmpl.c:sgr_5x5_c:
  830|  7.58k|{
  831|  7.58k|    ALIGN_STK_16(int32_t, sumsq_buf, BUF_STRIDE * 5 + 16,);
  ------------------
  |  |  100|  7.58k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  7.58k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  832|  7.58k|    ALIGN_STK_16(coef, sum_buf, BUF_STRIDE * 5 + 16,);
  ------------------
  |  |  100|  7.58k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  7.58k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  833|  7.58k|    int32_t *sumsq_ptrs[5], *sumsq_rows[5];
  834|  7.58k|    coef *sum_ptrs[5], *sum_rows[5];
  835|  45.5k|    for (int i = 0; i < 5; i++) {
  ------------------
  |  Branch (835:21): [True: 37.9k, False: 7.58k]
  ------------------
  836|  37.9k|        sumsq_rows[i] = &sumsq_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  37.9k|#define BUF_STRIDE (384 + 16)
  ------------------
  837|  37.9k|        sum_rows[i] = &sum_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  37.9k|#define BUF_STRIDE (384 + 16)
  ------------------
  838|  37.9k|    }
  839|       |
  840|  7.58k|    ALIGN_STK_16(int32_t, A_buf, BUF_STRIDE * 2 + 16,);
  ------------------
  |  |  100|  7.58k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  7.58k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  841|  7.58k|    ALIGN_STK_16(coef, B_buf, BUF_STRIDE * 2 + 16,);
  ------------------
  |  |  100|  7.58k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  7.58k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  842|  7.58k|    int32_t *A_ptrs[2];
  843|  7.58k|    coef *B_ptrs[2];
  844|  22.7k|    for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (844:21): [True: 15.1k, False: 7.58k]
  ------------------
  845|  15.1k|        A_ptrs[i] = &A_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  15.1k|#define BUF_STRIDE (384 + 16)
  ------------------
  846|  15.1k|        B_ptrs[i] = &B_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  15.1k|#define BUF_STRIDE (384 + 16)
  ------------------
  847|  15.1k|    }
  848|  7.58k|    const pixel *src = dst;
  849|  7.58k|    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
  ------------------
  |  |   53|  7.58k|#define PXSTRIDE(x) (x)
  ------------------
  850|       |
  851|  7.58k|    if (edges & LR_HAVE_TOP) {
  ------------------
  |  Branch (851:9): [True: 4.90k, False: 2.68k]
  ------------------
  852|  4.90k|        sumsq_ptrs[0] = sumsq_rows[0];
  853|  4.90k|        sumsq_ptrs[1] = sumsq_rows[0];
  854|  4.90k|        sumsq_ptrs[2] = sumsq_rows[1];
  855|  4.90k|        sumsq_ptrs[3] = sumsq_rows[2];
  856|  4.90k|        sumsq_ptrs[4] = sumsq_rows[3];
  857|  4.90k|        sum_ptrs[0] = sum_rows[0];
  858|  4.90k|        sum_ptrs[1] = sum_rows[0];
  859|  4.90k|        sum_ptrs[2] = sum_rows[1];
  860|  4.90k|        sum_ptrs[3] = sum_rows[2];
  861|  4.90k|        sum_ptrs[4] = sum_rows[3];
  862|       |
  863|  4.90k|        sgr_box5_row_h(sumsq_rows[0], sum_rows[0], NULL, lpf, w, edges);
  864|  4.90k|        lpf += PXSTRIDE(stride);
  ------------------
  |  |   53|  4.90k|#define PXSTRIDE(x) (x)
  ------------------
  865|  4.90k|        sgr_box5_row_h(sumsq_rows[1], sum_rows[1], NULL, lpf, w, edges);
  866|       |
  867|  4.90k|        sgr_box5_row_h(sumsq_rows[2], sum_rows[2], left, src, w, edges);
  868|  4.90k|        left++;
  869|  4.90k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  4.90k|#define PXSTRIDE(x) (x)
  ------------------
  870|       |
  871|  4.90k|        if (--h <= 0)
  ------------------
  |  Branch (871:13): [True: 605, False: 4.30k]
  ------------------
  872|    605|            goto vert_1;
  873|       |
  874|  4.30k|        sgr_box5_row_h(sumsq_rows[3], sum_rows[3], left, src, w, edges);
  875|  4.30k|        left++;
  876|  4.30k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  4.30k|#define PXSTRIDE(x) (x)
  ------------------
  877|  4.30k|        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  878|  4.30k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  4.30k|#define BITDEPTH_MAX 0xff
  ------------------
  879|  4.30k|        rotate(A_ptrs, B_ptrs, 2);
  880|       |
  881|  4.30k|        if (--h <= 0)
  ------------------
  |  Branch (881:13): [True: 417, False: 3.88k]
  ------------------
  882|    417|            goto vert_2;
  883|       |
  884|       |        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
  885|       |        // one of them to point at the previously unused rows[4].
  886|  3.88k|        sumsq_ptrs[3] = sumsq_rows[4];
  887|  3.88k|        sum_ptrs[3] = sum_rows[4];
  888|  3.88k|    } else {
  889|  2.68k|        sumsq_ptrs[0] = sumsq_rows[0];
  890|  2.68k|        sumsq_ptrs[1] = sumsq_rows[0];
  891|  2.68k|        sumsq_ptrs[2] = sumsq_rows[0];
  892|  2.68k|        sumsq_ptrs[3] = sumsq_rows[0];
  893|  2.68k|        sumsq_ptrs[4] = sumsq_rows[0];
  894|  2.68k|        sum_ptrs[0] = sum_rows[0];
  895|  2.68k|        sum_ptrs[1] = sum_rows[0];
  896|  2.68k|        sum_ptrs[2] = sum_rows[0];
  897|  2.68k|        sum_ptrs[3] = sum_rows[0];
  898|  2.68k|        sum_ptrs[4] = sum_rows[0];
  899|       |
  900|  2.68k|        sgr_box5_row_h(sumsq_rows[0], sum_rows[0], left, src, w, edges);
  901|  2.68k|        left++;
  902|  2.68k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.68k|#define PXSTRIDE(x) (x)
  ------------------
  903|       |
  904|  2.68k|        if (--h <= 0)
  ------------------
  |  Branch (904:13): [True: 169, False: 2.51k]
  ------------------
  905|    169|            goto vert_1;
  906|       |
  907|  2.51k|        sumsq_ptrs[4] = sumsq_rows[1];
  908|  2.51k|        sum_ptrs[4] = sum_rows[1];
  909|       |
  910|  2.51k|        sgr_box5_row_h(sumsq_rows[1], sum_rows[1], left, src, w, edges);
  911|  2.51k|        left++;
  912|  2.51k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.51k|#define PXSTRIDE(x) (x)
  ------------------
  913|       |
  914|  2.51k|        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  915|  2.51k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.51k|#define BITDEPTH_MAX 0xff
  ------------------
  916|  2.51k|        rotate(A_ptrs, B_ptrs, 2);
  917|       |
  918|  2.51k|        if (--h <= 0)
  ------------------
  |  Branch (918:13): [True: 180, False: 2.33k]
  ------------------
  919|    180|            goto vert_2;
  920|       |
  921|  2.33k|        sumsq_ptrs[3] = sumsq_rows[2];
  922|  2.33k|        sumsq_ptrs[4] = sumsq_rows[3];
  923|  2.33k|        sum_ptrs[3] = sum_rows[2];
  924|  2.33k|        sum_ptrs[4] = sum_rows[3];
  925|       |
  926|  2.33k|        sgr_box5_row_h(sumsq_rows[2], sum_rows[2], left, src, w, edges);
  927|  2.33k|        left++;
  928|  2.33k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.33k|#define PXSTRIDE(x) (x)
  ------------------
  929|       |
  930|  2.33k|        if (--h <= 0)
  ------------------
  |  Branch (930:13): [True: 172, False: 2.15k]
  ------------------
  931|    172|            goto odd;
  932|       |
  933|  2.15k|        sgr_box5_row_h(sumsq_rows[3], sum_rows[3], left, src, w, edges);
  934|  2.15k|        left++;
  935|  2.15k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.15k|#define PXSTRIDE(x) (x)
  ------------------
  936|       |
  937|  2.15k|        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  938|  2.15k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.15k|#define BITDEPTH_MAX 0xff
  ------------------
  939|  2.15k|        sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
  940|  2.15k|                    w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
  941|       |
  942|  2.15k|        if (--h <= 0)
  ------------------
  |  Branch (942:13): [True: 93, False: 2.06k]
  ------------------
  943|     93|            goto vert_2;
  944|       |
  945|       |        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
  946|       |        // one of them to point at the previously unused rows[4].
  947|  2.06k|        sumsq_ptrs[3] = sumsq_rows[4];
  948|  2.06k|        sum_ptrs[3] = sum_rows[4];
  949|  2.06k|    }
  950|       |
  951|   151k|    do {
  952|   151k|        sgr_box5_row_h(sumsq_ptrs[3], sum_ptrs[3], left, src, w, edges);
  953|   151k|        left++;
  954|   151k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|   151k|#define PXSTRIDE(x) (x)
  ------------------
  955|       |
  956|   151k|        if (--h <= 0)
  ------------------
  |  Branch (956:13): [True: 705, False: 150k]
  ------------------
  957|    705|            goto odd;
  958|       |
  959|   150k|        sgr_box5_row_h(sumsq_ptrs[4], sum_ptrs[4], left, src, w, edges);
  960|   150k|        left++;
  961|   150k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|   150k|#define PXSTRIDE(x) (x)
  ------------------
  962|       |
  963|   150k|        sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  964|   150k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|   150k|#define BITDEPTH_MAX 0xff
  ------------------
  965|   150k|        sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
  966|   150k|                    w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
  967|   150k|    } while (--h > 0);
  ------------------
  |  Branch (967:14): [True: 145k, False: 5.24k]
  ------------------
  968|       |
  969|  5.24k|    if (!(edges & LR_HAVE_BOTTOM))
  ------------------
  |  Branch (969:9): [True: 306, False: 4.93k]
  ------------------
  970|    306|        goto vert_2;
  971|       |
  972|  4.93k|    sgr_box5_row_h(sumsq_ptrs[3], sum_ptrs[3], NULL, lpf_bottom, w, edges);
  973|  4.93k|    lpf_bottom += PXSTRIDE(stride);
  ------------------
  |  |   53|  4.93k|#define PXSTRIDE(x) (x)
  ------------------
  974|  4.93k|    sgr_box5_row_h(sumsq_ptrs[4], sum_ptrs[4], NULL, lpf_bottom, w, edges);
  975|       |
  976|  5.93k|output_2:
  977|  5.93k|    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  978|  5.93k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  5.93k|#define BITDEPTH_MAX 0xff
  ------------------
  979|  5.93k|    sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
  980|  5.93k|                w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
  981|  5.93k|    return;
  982|       |
  983|    996|vert_2:
  984|       |    // Duplicate the last row twice more
  985|    996|    sumsq_ptrs[3] = sumsq_ptrs[2];
  986|    996|    sumsq_ptrs[4] = sumsq_ptrs[2];
  987|    996|    sum_ptrs[3] = sum_ptrs[2];
  988|    996|    sum_ptrs[4] = sum_ptrs[2];
  989|    996|    goto output_2;
  990|       |
  991|    877|odd:
  992|       |    // Copy the last row as padding once
  993|    877|    sumsq_ptrs[4] = sumsq_ptrs[3];
  994|    877|    sum_ptrs[4] = sum_ptrs[3];
  995|       |
  996|    877|    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
  997|    877|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|    877|#define BITDEPTH_MAX 0xff
  ------------------
  998|    877|    sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
  999|    877|                w, 2, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
 1000|       |
 1001|  1.65k|output_1:
 1002|       |    // Duplicate the last row twice more
 1003|  1.65k|    sumsq_ptrs[3] = sumsq_ptrs[2];
 1004|  1.65k|    sumsq_ptrs[4] = sumsq_ptrs[2];
 1005|  1.65k|    sum_ptrs[3] = sum_ptrs[2];
 1006|  1.65k|    sum_ptrs[4] = sum_ptrs[2];
 1007|       |
 1008|  1.65k|    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
 1009|  1.65k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.65k|#define BITDEPTH_MAX 0xff
  ------------------
 1010|       |    // Output only one row
 1011|  1.65k|    sgr_finish2(&dst, stride, A_ptrs, B_ptrs,
 1012|  1.65k|                w, 1, params->sgr.w0 HIGHBD_TAIL_SUFFIX);
 1013|  1.65k|    return;
 1014|       |
 1015|    774|vert_1:
 1016|       |    // Copy the last row as padding once
 1017|    774|    sumsq_ptrs[4] = sumsq_ptrs[3];
 1018|    774|    sum_ptrs[4] = sum_ptrs[3];
 1019|       |
 1020|    774|    sgr_box5_vert(sumsq_ptrs, sum_ptrs, A_ptrs[1], B_ptrs[1],
 1021|    774|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|    774|#define BITDEPTH_MAX 0xff
  ------------------
 1022|    774|    rotate(A_ptrs, B_ptrs, 2);
 1023|       |
 1024|    774|    goto output_1;
 1025|    877|}
looprestoration_tmpl.c:sgr_box5_row_h:
  441|  1.48M|{
  442|  1.48M|    sumsq++;
  443|  1.48M|    sum++;
  444|  1.48M|    int a = edges & LR_HAVE_LEFT ? (left ? left[0][1] : src[-3]) : src[0];
  ------------------
  |  Branch (444:13): [True: 818k, False: 663k]
  |  Branch (444:37): [True: 769k, False: 48.6k]
  ------------------
  445|  1.48M|    int b = edges & LR_HAVE_LEFT ? (left ? left[0][2] : src[-2]) : src[0];
  ------------------
  |  Branch (445:13): [True: 818k, False: 663k]
  |  Branch (445:37): [True: 769k, False: 48.6k]
  ------------------
  446|  1.48M|    int c = edges & LR_HAVE_LEFT ? (left ? left[0][3] : src[-1]) : src[0];
  ------------------
  |  Branch (446:13): [True: 818k, False: 663k]
  |  Branch (446:37): [True: 769k, False: 48.6k]
  ------------------
  447|  1.48M|    int d = src[0];
  448|   139M|    for (int x = -1; x < w + 1; x++) {
  ------------------
  |  Branch (448:22): [True: 137M, False: 1.48M]
  ------------------
  449|   137M|        int e = (x + 2 < w || (edges & LR_HAVE_RIGHT)) ? src[x + 2] : src[w - 1];
  ------------------
  |  Branch (449:18): [True: 133M, False: 4.42M]
  |  Branch (449:31): [True: 2.47M, False: 1.95M]
  ------------------
  450|   137M|        sum[x] = a + b + c + d + e;
  451|   137M|        sumsq[x] = a * a + b * b + c * c + d * d + e * e;
  452|   137M|        a = b;
  453|   137M|        b = c;
  454|   137M|        c = d;
  455|   137M|        d = e;
  456|   137M|    }
  457|  1.48M|}
looprestoration_tmpl.c:sgr_box5_vert:
  537|   735k|{
  538|   735k|    sgr_box5_row_v(sumsq, sum, sumsq_out, sum_out, w);
  539|   735k|    sgr_calc_row_ab(sumsq_out, sum_out, w, s, bitdepth_max, 25, 164);
  540|   735k|    rotate5_x2(sumsq, sum);
  541|   735k|}
looprestoration_tmpl.c:sgr_box5_row_v:
  488|   735k|{
  489|  69.0M|    for (int x = 0; x < w + 2; x++) {
  ------------------
  |  Branch (489:21): [True: 68.3M, False: 735k]
  ------------------
  490|  68.3M|        int sq_a = sumsq[0][x];
  491|  68.3M|        int sq_b = sumsq[1][x];
  492|  68.3M|        int sq_c = sumsq[2][x];
  493|  68.3M|        int sq_d = sumsq[3][x];
  494|  68.3M|        int sq_e = sumsq[4][x];
  495|  68.3M|        int s_a = sum[0][x];
  496|  68.3M|        int s_b = sum[1][x];
  497|  68.3M|        int s_c = sum[2][x];
  498|  68.3M|        int s_d = sum[3][x];
  499|  68.3M|        int s_e = sum[4][x];
  500|  68.3M|        sumsq_out[x] = sq_a + sq_b + sq_c + sq_d + sq_e;
  501|  68.3M|        sum_out[x] = s_a + s_b + s_c + s_d + s_e;
  502|  68.3M|    }
  503|   735k|}
looprestoration_tmpl.c:sgr_calc_row_ab:
  507|  2.19M|{
  508|  2.19M|    const int bitdepth_min_8 = bitdepth_from_max(bitdepth_max) - 8;
  ------------------
  |  |   58|  2.19M|#define bitdepth_from_max(x) 8
  ------------------
  509|   188M|    for (int i = 0; i < w + 2; i++) {
  ------------------
  |  Branch (509:21): [True: 186M, False: 2.19M]
  ------------------
  510|   186M|        const int a =
  511|   186M|            (AA[i] + ((1 << (2 * bitdepth_min_8)) >> 1)) >> (2 * bitdepth_min_8);
  512|   186M|        const int b =
  513|   186M|            (BB[i] + ((1 << bitdepth_min_8) >> 1)) >> bitdepth_min_8;
  514|       |
  515|   186M|        const unsigned p = imax(a * n - b * b, 0);
  516|   186M|        const unsigned z = (p * s + (1 << 19)) >> 20;
  517|   186M|        const unsigned x = dav1d_sgr_x_by_x[umin(z, 255)];
  518|       |
  519|       |        // This is where we invert A and B, so that B is of size coef.
  520|   186M|        AA[i] = (x * BB[i] * sgr_one_by_x + (1 << 11)) >> 12;
  521|   186M|        BB[i] = x;
  522|   186M|    }
  523|  2.19M|}
looprestoration_tmpl.c:rotate5_x2:
  402|   732k|{
  403|   732k|    int32_t *tmp32[2];
  404|   732k|    coef *tmpc[2];
  405|  2.19M|    for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (405:21): [True: 1.46M, False: 732k]
  ------------------
  406|  1.46M|        tmp32[i] = sumsq_ptrs[i];
  407|  1.46M|        tmpc[i] = sum_ptrs[i];
  408|  1.46M|    }
  409|  2.92M|    for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (409:21): [True: 2.19M, False: 732k]
  ------------------
  410|  2.19M|        sumsq_ptrs[i] = sumsq_ptrs[i + 2];
  411|  2.19M|        sum_ptrs[i] = sum_ptrs[i + 2];
  412|  2.19M|    }
  413|  2.19M|    for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (413:21): [True: 1.46M, False: 732k]
  ------------------
  414|  1.46M|        sumsq_ptrs[3 + i] = tmp32[i];
  415|  1.46M|        sum_ptrs[3 + i] = tmpc[i];
  416|  1.46M|    }
  417|   732k|}
looprestoration_tmpl.c:rotate:
  390|  3.63M|{
  391|  3.63M|    int32_t *tmp32 = sumsq_ptrs[0];
  392|  3.63M|    coef *tmpc = sum_ptrs[0];
  393|  11.3M|    for (int i = 0; i < n - 1; i++) {
  ------------------
  |  Branch (393:21): [True: 7.66M, False: 3.63M]
  ------------------
  394|  7.66M|        sumsq_ptrs[i] = sumsq_ptrs[i + 1];
  395|  7.66M|        sum_ptrs[i] = sum_ptrs[i + 1];
  396|  7.66M|    }
  397|  3.63M|    sumsq_ptrs[n - 1] = tmp32;
  398|  3.63M|    sum_ptrs[n - 1] = tmpc;
  399|  3.63M|}
looprestoration_tmpl.c:sgr_finish2:
  645|   161k|{
  646|   161k|    ALIGN_STK_16(coef, tmp, 2*FILTER_OUT_STRIDE,);
  ------------------
  |  |  100|   161k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|   161k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  647|       |
  648|   161k|    sgr_finish_filter2(tmp, *dst, stride, A_ptrs, B_ptrs, w, h);
  649|   161k|    sgr_weighted_row1(*dst, tmp, w, w1 HIGHBD_TAIL_SUFFIX);
  650|   161k|    *dst += PXSTRIDE(stride);
  ------------------
  |  |   53|   161k|#define PXSTRIDE(x) (x)
  ------------------
  651|   161k|    if (h > 1) {
  ------------------
  |  Branch (651:9): [True: 159k, False: 1.75k]
  ------------------
  652|   159k|        sgr_weighted_row1(*dst, tmp + FILTER_OUT_STRIDE, w, w1 HIGHBD_TAIL_SUFFIX);
  ------------------
  |  |  572|   159k|#define FILTER_OUT_STRIDE (384)
  ------------------
  653|   159k|        *dst += PXSTRIDE(stride);
  ------------------
  |  |   53|   159k|#define PXSTRIDE(x) (x)
  ------------------
  654|   159k|    }
  655|   161k|    rotate(A_ptrs, B_ptrs, 2);
  656|   161k|}
looprestoration_tmpl.c:sgr_finish_filter2:
  579|   701k|{
  580|   701k|#define SIX_NEIGHBORS(P, i)\
  581|   701k|    ((P[0][i]     + P[1][i]) * 6 +   \
  582|   701k|     (P[0][i - 1] + P[1][i - 1] +    \
  583|   701k|      P[0][i + 1] + P[1][i + 1]) * 5)
  584|  64.4M|    for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (584:21): [True: 63.7M, False: 701k]
  ------------------
  585|  63.7M|        const int a = SIX_NEIGHBORS(B_ptrs, i + 1);
  ------------------
  |  |  581|  63.7M|    ((P[0][i]     + P[1][i]) * 6 +   \
  |  |  582|  63.7M|     (P[0][i - 1] + P[1][i - 1] +    \
  |  |  583|  63.7M|      P[0][i + 1] + P[1][i + 1]) * 5)
  ------------------
  586|  63.7M|        const int b = SIX_NEIGHBORS(A_ptrs, i + 1);
  ------------------
  |  |  581|  63.7M|    ((P[0][i]     + P[1][i]) * 6 +   \
  |  |  582|  63.7M|     (P[0][i - 1] + P[1][i - 1] +    \
  |  |  583|  63.7M|      P[0][i + 1] + P[1][i + 1]) * 5)
  ------------------
  587|  63.7M|        tmp[i] = (b - a * src[i] + (1 << 8)) >> 9;
  588|  63.7M|    }
  589|   701k|    if (h <= 1)
  ------------------
  |  Branch (589:9): [True: 7.02k, False: 694k]
  ------------------
  590|  7.02k|        return;
  591|   694k|    tmp += FILTER_OUT_STRIDE;
  ------------------
  |  |  572|   694k|#define FILTER_OUT_STRIDE (384)
  ------------------
  592|   694k|    src += PXSTRIDE(src_stride);
  ------------------
  |  |   53|   694k|#define PXSTRIDE(x) (x)
  ------------------
  593|   694k|    const int32_t *A = &A_ptrs[1][1];
  594|   694k|    const coef *B = &B_ptrs[1][1];
  595|  64.8M|    for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (595:21): [True: 64.1M, False: 694k]
  ------------------
  596|  64.1M|        const int a = B[i] * 6 + (B[i - 1] + B[i + 1]) * 5;
  597|  64.1M|        const int b = A[i] * 6 + (A[i - 1] + A[i + 1]) * 5;
  598|  64.1M|        tmp[i] = (b - a * src[i] + (1 << 7)) >> 8;
  599|  64.1M|    }
  600|   694k|#undef SIX_NEIGHBORS
  601|   694k|}
looprestoration_tmpl.c:sgr_weighted_row1:
  605|   639k|{
  606|  72.8M|    for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (606:21): [True: 72.2M, False: 639k]
  ------------------
  607|  72.2M|        const int v = w1 * t1[i];
  608|  72.2M|        dst[i] = iclip_pixel(dst[i] + ((v + (1 << 10)) >> 11));
  ------------------
  |  |   49|  72.2M|#define iclip_pixel iclip_u8
  ------------------
  609|  72.2M|    }
  610|   639k|}
looprestoration_tmpl.c:sgr_3x3_c:
  684|  7.69k|{
  685|  7.69k|#define BUF_STRIDE (384 + 16)
  686|  7.69k|    ALIGN_STK_16(int32_t, sumsq_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  7.69k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  7.69k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  687|  7.69k|    ALIGN_STK_16(coef, sum_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  7.69k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  7.69k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  688|  7.69k|    int32_t *sumsq_ptrs[3], *sumsq_rows[3];
  689|  7.69k|    coef *sum_ptrs[3], *sum_rows[3];
  690|  30.7k|    for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (690:21): [True: 23.0k, False: 7.69k]
  ------------------
  691|  23.0k|        sumsq_rows[i] = &sumsq_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  23.0k|#define BUF_STRIDE (384 + 16)
  ------------------
  692|  23.0k|        sum_rows[i] = &sum_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  23.0k|#define BUF_STRIDE (384 + 16)
  ------------------
  693|  23.0k|    }
  694|       |
  695|  7.69k|    ALIGN_STK_16(int32_t, A_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  7.69k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  7.69k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  696|  7.69k|    ALIGN_STK_16(coef, B_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  7.69k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  7.69k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  697|  7.69k|    int32_t *A_ptrs[3];
  698|  7.69k|    coef *B_ptrs[3];
  699|  30.7k|    for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (699:21): [True: 23.0k, False: 7.69k]
  ------------------
  700|  23.0k|        A_ptrs[i] = &A_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  23.0k|#define BUF_STRIDE (384 + 16)
  ------------------
  701|  23.0k|        B_ptrs[i] = &B_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  23.0k|#define BUF_STRIDE (384 + 16)
  ------------------
  702|  23.0k|    }
  703|  7.69k|    const pixel *src = dst;
  704|  7.69k|    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
  ------------------
  |  |   53|  7.69k|#define PXSTRIDE(x) (x)
  ------------------
  705|       |
  706|  7.69k|    if (edges & LR_HAVE_TOP) {
  ------------------
  |  Branch (706:9): [True: 5.13k, False: 2.55k]
  ------------------
  707|  5.13k|        sumsq_ptrs[0] = sumsq_rows[0];
  708|  5.13k|        sumsq_ptrs[1] = sumsq_rows[1];
  709|  5.13k|        sumsq_ptrs[2] = sumsq_rows[2];
  710|  5.13k|        sum_ptrs[0] = sum_rows[0];
  711|  5.13k|        sum_ptrs[1] = sum_rows[1];
  712|  5.13k|        sum_ptrs[2] = sum_rows[2];
  713|       |
  714|  5.13k|        sgr_box3_row_h(sumsq_rows[0], sum_rows[0], NULL, lpf, w, edges);
  715|  5.13k|        lpf += PXSTRIDE(stride);
  ------------------
  |  |   53|  5.13k|#define PXSTRIDE(x) (x)
  ------------------
  716|  5.13k|        sgr_box3_row_h(sumsq_rows[1], sum_rows[1], NULL, lpf, w, edges);
  717|       |
  718|  5.13k|        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  719|  5.13k|                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  5.13k|#define BITDEPTH_MAX 0xff
  ------------------
  720|  5.13k|        left++;
  721|  5.13k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  5.13k|#define PXSTRIDE(x) (x)
  ------------------
  722|  5.13k|        rotate(A_ptrs, B_ptrs, 3);
  723|       |
  724|  5.13k|        if (--h <= 0)
  ------------------
  |  Branch (724:13): [True: 757, False: 4.37k]
  ------------------
  725|    757|            goto vert_1;
  726|       |
  727|  4.37k|        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  728|  4.37k|                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  4.37k|#define BITDEPTH_MAX 0xff
  ------------------
  729|  4.37k|        left++;
  730|  4.37k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  4.37k|#define PXSTRIDE(x) (x)
  ------------------
  731|  4.37k|        rotate(A_ptrs, B_ptrs, 3);
  732|       |
  733|  4.37k|        if (--h <= 0)
  ------------------
  |  Branch (733:13): [True: 947, False: 3.43k]
  ------------------
  734|    947|            goto vert_2;
  735|  4.37k|    } else {
  736|  2.55k|        sumsq_ptrs[0] = sumsq_rows[0];
  737|  2.55k|        sumsq_ptrs[1] = sumsq_rows[0];
  738|  2.55k|        sumsq_ptrs[2] = sumsq_rows[0];
  739|  2.55k|        sum_ptrs[0] = sum_rows[0];
  740|  2.55k|        sum_ptrs[1] = sum_rows[0];
  741|  2.55k|        sum_ptrs[2] = sum_rows[0];
  742|       |
  743|  2.55k|        sgr_box3_row_h(sumsq_rows[0], sum_rows[0], left, src, w, edges);
  744|  2.55k|        left++;
  745|  2.55k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.55k|#define PXSTRIDE(x) (x)
  ------------------
  746|       |
  747|  2.55k|        sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  748|  2.55k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.55k|#define BITDEPTH_MAX 0xff
  ------------------
  749|  2.55k|        rotate(A_ptrs, B_ptrs, 3);
  750|       |
  751|  2.55k|        if (--h <= 0)
  ------------------
  |  Branch (751:13): [True: 128, False: 2.43k]
  ------------------
  752|    128|            goto vert_1;
  753|       |
  754|  2.43k|        sumsq_ptrs[2] = sumsq_rows[1];
  755|  2.43k|        sum_ptrs[2] = sum_rows[1];
  756|       |
  757|  2.43k|        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  758|  2.43k|                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.43k|#define BITDEPTH_MAX 0xff
  ------------------
  759|  2.43k|        left++;
  760|  2.43k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  2.43k|#define PXSTRIDE(x) (x)
  ------------------
  761|  2.43k|        rotate(A_ptrs, B_ptrs, 3);
  762|       |
  763|  2.43k|        if (--h <= 0)
  ------------------
  |  Branch (763:13): [True: 97, False: 2.33k]
  ------------------
  764|     97|            goto vert_2;
  765|       |
  766|  2.33k|        sumsq_ptrs[2] = sumsq_rows[2];
  767|  2.33k|        sum_ptrs[2] = sum_rows[2];
  768|  2.33k|    }
  769|       |
  770|   305k|    do {
  771|   305k|        sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  772|   305k|                    left, src, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|   305k|#define BITDEPTH_MAX 0xff
  ------------------
  773|   305k|        left++;
  774|   305k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|   305k|#define PXSTRIDE(x) (x)
  ------------------
  775|       |
  776|   305k|        sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  777|   305k|                    w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  778|   305k|    } while (--h > 0);
  ------------------
  |  Branch (778:14): [True: 299k, False: 5.76k]
  ------------------
  779|       |
  780|  5.76k|    if (!(edges & LR_HAVE_BOTTOM))
  ------------------
  |  Branch (780:9): [True: 606, False: 5.15k]
  ------------------
  781|    606|        goto vert_2;
  782|       |
  783|  5.15k|    sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  784|  5.15k|                NULL, lpf_bottom, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  5.15k|#define BITDEPTH_MAX 0xff
  ------------------
  785|  5.15k|    lpf_bottom += PXSTRIDE(stride);
  ------------------
  |  |   53|  5.15k|#define PXSTRIDE(x) (x)
  ------------------
  786|       |
  787|  5.15k|    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  788|  5.15k|                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  789|       |
  790|  5.15k|    sgr_box3_hv(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  791|  5.15k|                NULL, lpf_bottom, w, params->sgr.s1, edges, BITDEPTH_MAX);
  ------------------
  |  |   59|  5.15k|#define BITDEPTH_MAX 0xff
  ------------------
  792|       |
  793|  5.15k|    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  794|  5.15k|                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  795|  5.15k|    return;
  796|       |
  797|  1.65k|vert_2:
  798|  1.65k|    sumsq_ptrs[2] = sumsq_ptrs[1];
  799|  1.65k|    sum_ptrs[2] = sum_ptrs[1];
  800|  1.65k|    sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  801|  1.65k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.65k|#define BITDEPTH_MAX 0xff
  ------------------
  802|       |
  803|  1.65k|    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  804|  1.65k|                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  805|       |
  806|  2.53k|output_1:
  807|  2.53k|    sumsq_ptrs[2] = sumsq_ptrs[1];
  808|  2.53k|    sum_ptrs[2] = sum_ptrs[1];
  809|  2.53k|    sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  810|  2.53k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.53k|#define BITDEPTH_MAX 0xff
  ------------------
  811|       |
  812|  2.53k|    sgr_finish1(&dst, stride, A_ptrs, B_ptrs,
  813|  2.53k|                w, params->sgr.w1 HIGHBD_TAIL_SUFFIX);
  814|  2.53k|    return;
  815|       |
  816|    885|vert_1:
  817|    885|    sumsq_ptrs[2] = sumsq_ptrs[1];
  818|    885|    sum_ptrs[2] = sum_ptrs[1];
  819|    885|    sgr_box3_vert(sumsq_ptrs, sum_ptrs, A_ptrs[2], B_ptrs[2],
  820|    885|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|    885|#define BITDEPTH_MAX 0xff
  ------------------
  821|    885|    rotate(A_ptrs, B_ptrs, 3);
  822|    885|    goto output_1;
  823|  1.65k|}
looprestoration_tmpl.c:sgr_box3_row_h:
  423|  1.47M|{
  424|  1.47M|    sumsq++;
  425|  1.47M|    sum++;
  426|  1.47M|    int a = edges & LR_HAVE_LEFT ? (left ? left[0][2] : src[-2]) : src[0];
  ------------------
  |  Branch (426:13): [True: 914k, False: 562k]
  |  Branch (426:37): [True: 859k, False: 54.7k]
  ------------------
  427|  1.47M|    int b = edges & LR_HAVE_LEFT ? (left ? left[0][3] : src[-1]) : src[0];
  ------------------
  |  Branch (427:13): [True: 914k, False: 562k]
  |  Branch (427:37): [True: 859k, False: 54.8k]
  ------------------
  428|   147M|    for (int x = -1; x < w + 1; x++) {
  ------------------
  |  Branch (428:22): [True: 146M, False: 1.47M]
  ------------------
  429|   146M|        int c = (x + 1 < w || (edges & LR_HAVE_RIGHT)) ? src[x + 1] : src[w - 1];
  ------------------
  |  Branch (429:18): [True: 143M, False: 2.95M]
  |  Branch (429:31): [True: 1.82M, False: 1.13M]
  ------------------
  430|   146M|        sum[x] = a + b + c;
  431|   146M|        sumsq[x] = a * a + b * b + c * c;
  432|   146M|        a = b;
  433|   146M|        b = c;
  434|   146M|    }
  435|  1.47M|}
looprestoration_tmpl.c:sgr_box3_hv:
  550|   327k|{
  551|   327k|    sgr_box3_row_h(sumsq[2], sum[2], left, src, w, edges);
  552|   327k|    sgr_box3_vert(sumsq, sum, AA, BB, w, s, bitdepth_max);
  553|   327k|}
looprestoration_tmpl.c:sgr_box3_vert:
  528|  1.45M|{
  529|  1.45M|    sgr_box3_row_v(sumsq, sum, sumsq_out, sum_out, w);
  530|  1.45M|    sgr_calc_row_ab(sumsq_out, sum_out, w, s, bitdepth_max, 9, 455);
  531|  1.45M|    rotate(sumsq, sum, 3);
  532|  1.45M|}
looprestoration_tmpl.c:sgr_box3_row_v:
  472|  1.45M|{
  473|   144M|    for (int x = 0; x < w + 2; x++) {
  ------------------
  |  Branch (473:21): [True: 142M, False: 1.45M]
  ------------------
  474|   142M|        int sq_a = sumsq[0][x];
  475|   142M|        int sq_b = sumsq[1][x];
  476|   142M|        int sq_c = sumsq[2][x];
  477|   142M|        int s_a = sum[0][x];
  478|   142M|        int s_b = sum[1][x];
  479|   142M|        int s_c = sum[2][x];
  480|   142M|        sumsq_out[x] = sq_a + sq_b + sq_c;
  481|   142M|        sum_out[x] = s_a + s_b + s_c;
  482|   142M|    }
  483|  1.45M|}
looprestoration_tmpl.c:sgr_finish1:
  631|   319k|{
  632|       |    // Only one single row, no stride needed
  633|   319k|    ALIGN_STK_16(coef, tmp, 384,);
  ------------------
  |  |  100|   319k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|   319k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  634|       |
  635|   319k|    sgr_finish_filter_row1(tmp, *dst, A_ptrs, B_ptrs, w);
  636|   319k|    sgr_weighted_row1(*dst, tmp, w, w1 HIGHBD_TAIL_SUFFIX);
  637|   319k|    *dst += PXSTRIDE(stride);
  ------------------
  |  |   53|   319k|#define PXSTRIDE(x) (x)
  ------------------
  638|   319k|    rotate(A_ptrs, B_ptrs, 3);
  639|   319k|}
looprestoration_tmpl.c:sgr_finish_filter_row1:
  559|  1.39M|{
  560|  1.39M|#define EIGHT_NEIGHBORS(P, i)\
  561|  1.39M|    ((P[1][i] + P[1][i - 1] + P[1][i + 1] + P[0][i] + P[2][i]) * 4 + \
  562|  1.39M|     (P[0][i - 1] + P[2][i - 1] +                           \
  563|  1.39M|      P[0][i + 1] + P[2][i + 1]) * 3)
  564|   132M|    for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (564:21): [True: 130M, False: 1.39M]
  ------------------
  565|   130M|        const int a = EIGHT_NEIGHBORS(B_ptrs, i + 1);
  ------------------
  |  |  561|   130M|    ((P[1][i] + P[1][i - 1] + P[1][i + 1] + P[0][i] + P[2][i]) * 4 + \
  |  |  562|   130M|     (P[0][i - 1] + P[2][i - 1] +                           \
  |  |  563|   130M|      P[0][i + 1] + P[2][i + 1]) * 3)
  ------------------
  566|   130M|        const int b = EIGHT_NEIGHBORS(A_ptrs, i + 1);
  ------------------
  |  |  561|   130M|    ((P[1][i] + P[1][i - 1] + P[1][i + 1] + P[0][i] + P[2][i]) * 4 + \
  |  |  562|   130M|     (P[0][i - 1] + P[2][i - 1] +                           \
  |  |  563|   130M|      P[0][i + 1] + P[2][i + 1]) * 3)
  ------------------
  567|   130M|        tmp[i] = (b - a * src[i] + (1 << 8)) >> 9;
  568|   130M|    }
  569|  1.39M|#undef EIGHT_NEIGHBORS
  570|  1.39M|}
looprestoration_tmpl.c:sgr_mix_c:
 1032|  24.1k|{
 1033|  24.1k|    ALIGN_STK_16(int32_t, sumsq5_buf, BUF_STRIDE * 5 + 16,);
  ------------------
  |  |  100|  24.1k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  24.1k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1034|  24.1k|    ALIGN_STK_16(coef, sum5_buf, BUF_STRIDE * 5 + 16,);
  ------------------
  |  |  100|  24.1k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  24.1k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1035|  24.1k|    int32_t *sumsq5_ptrs[5], *sumsq5_rows[5];
 1036|  24.1k|    coef *sum5_ptrs[5], *sum5_rows[5];
 1037|   145k|    for (int i = 0; i < 5; i++) {
  ------------------
  |  Branch (1037:21): [True: 120k, False: 24.1k]
  ------------------
 1038|   120k|        sumsq5_rows[i] = &sumsq5_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|   120k|#define BUF_STRIDE (384 + 16)
  ------------------
 1039|   120k|        sum5_rows[i] = &sum5_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|   120k|#define BUF_STRIDE (384 + 16)
  ------------------
 1040|   120k|    }
 1041|  24.1k|    ALIGN_STK_16(int32_t, sumsq3_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  24.1k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  24.1k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1042|  24.1k|    ALIGN_STK_16(coef, sum3_buf, BUF_STRIDE * 3 + 16,);
  ------------------
  |  |  100|  24.1k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  24.1k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1043|  24.1k|    int32_t *sumsq3_ptrs[3], *sumsq3_rows[3];
 1044|  24.1k|    coef *sum3_ptrs[3], *sum3_rows[3];
 1045|  96.7k|    for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (1045:21): [True: 72.5k, False: 24.1k]
  ------------------
 1046|  72.5k|        sumsq3_rows[i] = &sumsq3_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  72.5k|#define BUF_STRIDE (384 + 16)
  ------------------
 1047|  72.5k|        sum3_rows[i] = &sum3_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  72.5k|#define BUF_STRIDE (384 + 16)
  ------------------
 1048|  72.5k|    }
 1049|       |
 1050|  24.1k|    ALIGN_STK_16(int32_t, A5_buf, BUF_STRIDE * 2 + 16,);
  ------------------
  |  |  100|  24.1k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  24.1k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1051|  24.1k|    ALIGN_STK_16(coef, B5_buf, BUF_STRIDE * 2 + 16,);
  ------------------
  |  |  100|  24.1k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  24.1k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1052|  24.1k|    int32_t *A5_ptrs[2];
 1053|  24.1k|    coef *B5_ptrs[2];
 1054|  72.5k|    for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1054:21): [True: 48.3k, False: 24.1k]
  ------------------
 1055|  48.3k|        A5_ptrs[i] = &A5_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  48.3k|#define BUF_STRIDE (384 + 16)
  ------------------
 1056|  48.3k|        B5_ptrs[i] = &B5_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  48.3k|#define BUF_STRIDE (384 + 16)
  ------------------
 1057|  48.3k|    }
 1058|  24.1k|    ALIGN_STK_16(int32_t, A3_buf, BUF_STRIDE * 4 + 16,);
  ------------------
  |  |  100|  24.1k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  24.1k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1059|  24.1k|    ALIGN_STK_16(coef, B3_buf, BUF_STRIDE * 4 + 16,);
  ------------------
  |  |  100|  24.1k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|  24.1k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
 1060|  24.1k|    int32_t *A3_ptrs[4];
 1061|  24.1k|    coef *B3_ptrs[4];
 1062|   120k|    for (int i = 0; i < 4; i++) {
  ------------------
  |  Branch (1062:21): [True: 96.7k, False: 24.1k]
  ------------------
 1063|  96.7k|        A3_ptrs[i] = &A3_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  96.7k|#define BUF_STRIDE (384 + 16)
  ------------------
 1064|  96.7k|        B3_ptrs[i] = &B3_buf[i * BUF_STRIDE];
  ------------------
  |  |  685|  96.7k|#define BUF_STRIDE (384 + 16)
  ------------------
 1065|  96.7k|    }
 1066|  24.1k|    const pixel *src = dst;
 1067|  24.1k|    const pixel *lpf_bottom = lpf + 6*PXSTRIDE(stride);
  ------------------
  |  |   53|  24.1k|#define PXSTRIDE(x) (x)
  ------------------
 1068|       |
 1069|  24.1k|    if (edges & LR_HAVE_TOP) {
  ------------------
  |  Branch (1069:9): [True: 16.1k, False: 8.03k]
  ------------------
 1070|  16.1k|        sumsq5_ptrs[0] = sumsq5_rows[0];
 1071|  16.1k|        sumsq5_ptrs[1] = sumsq5_rows[0];
 1072|  16.1k|        sumsq5_ptrs[2] = sumsq5_rows[1];
 1073|  16.1k|        sumsq5_ptrs[3] = sumsq5_rows[2];
 1074|  16.1k|        sumsq5_ptrs[4] = sumsq5_rows[3];
 1075|  16.1k|        sum5_ptrs[0] = sum5_rows[0];
 1076|  16.1k|        sum5_ptrs[1] = sum5_rows[0];
 1077|  16.1k|        sum5_ptrs[2] = sum5_rows[1];
 1078|  16.1k|        sum5_ptrs[3] = sum5_rows[2];
 1079|  16.1k|        sum5_ptrs[4] = sum5_rows[3];
 1080|       |
 1081|  16.1k|        sumsq3_ptrs[0] = sumsq3_rows[0];
 1082|  16.1k|        sumsq3_ptrs[1] = sumsq3_rows[1];
 1083|  16.1k|        sumsq3_ptrs[2] = sumsq3_rows[2];
 1084|  16.1k|        sum3_ptrs[0] = sum3_rows[0];
 1085|  16.1k|        sum3_ptrs[1] = sum3_rows[1];
 1086|  16.1k|        sum3_ptrs[2] = sum3_rows[2];
 1087|       |
 1088|  16.1k|        sgr_box35_row_h(sumsq3_rows[0], sum3_rows[0],
 1089|  16.1k|                        sumsq5_rows[0], sum5_rows[0],
 1090|  16.1k|                        NULL, lpf, w, edges);
 1091|  16.1k|        lpf += PXSTRIDE(stride);
  ------------------
  |  |   53|  16.1k|#define PXSTRIDE(x) (x)
  ------------------
 1092|  16.1k|        sgr_box35_row_h(sumsq3_rows[1], sum3_rows[1],
 1093|  16.1k|                        sumsq5_rows[1], sum5_rows[1],
 1094|  16.1k|                        NULL, lpf, w, edges);
 1095|       |
 1096|  16.1k|        sgr_box35_row_h(sumsq3_rows[2], sum3_rows[2],
 1097|  16.1k|                        sumsq5_rows[2], sum5_rows[2],
 1098|  16.1k|                        left, src, w, edges);
 1099|  16.1k|        left++;
 1100|  16.1k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  16.1k|#define PXSTRIDE(x) (x)
  ------------------
 1101|       |
 1102|  16.1k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1103|  16.1k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  16.1k|#define BITDEPTH_MAX 0xff
  ------------------
 1104|  16.1k|        rotate(A3_ptrs, B3_ptrs, 4);
 1105|       |
 1106|  16.1k|        if (--h <= 0)
  ------------------
  |  Branch (1106:13): [True: 2.09k, False: 14.0k]
  ------------------
 1107|  2.09k|            goto vert_1;
 1108|       |
 1109|  14.0k|        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1110|  14.0k|                        sumsq5_rows[3], sum5_rows[3],
 1111|  14.0k|                        left, src, w, edges);
 1112|  14.0k|        left++;
 1113|  14.0k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  14.0k|#define PXSTRIDE(x) (x)
  ------------------
 1114|  14.0k|        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1115|  14.0k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  14.0k|#define BITDEPTH_MAX 0xff
  ------------------
 1116|  14.0k|        rotate(A5_ptrs, B5_ptrs, 2);
 1117|  14.0k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1118|  14.0k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  14.0k|#define BITDEPTH_MAX 0xff
  ------------------
 1119|  14.0k|        rotate(A3_ptrs, B3_ptrs, 4);
 1120|       |
 1121|  14.0k|        if (--h <= 0)
  ------------------
  |  Branch (1121:13): [True: 1.68k, False: 12.3k]
  ------------------
 1122|  1.68k|            goto vert_2;
 1123|       |
 1124|       |        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
 1125|       |        // one of them to point at the previously unused rows[4].
 1126|  12.3k|        sumsq5_ptrs[3] = sumsq5_rows[4];
 1127|  12.3k|        sum5_ptrs[3] = sum5_rows[4];
 1128|  12.3k|    } else {
 1129|  8.03k|        sumsq5_ptrs[0] = sumsq5_rows[0];
 1130|  8.03k|        sumsq5_ptrs[1] = sumsq5_rows[0];
 1131|  8.03k|        sumsq5_ptrs[2] = sumsq5_rows[0];
 1132|  8.03k|        sumsq5_ptrs[3] = sumsq5_rows[0];
 1133|  8.03k|        sumsq5_ptrs[4] = sumsq5_rows[0];
 1134|  8.03k|        sum5_ptrs[0] = sum5_rows[0];
 1135|  8.03k|        sum5_ptrs[1] = sum5_rows[0];
 1136|  8.03k|        sum5_ptrs[2] = sum5_rows[0];
 1137|  8.03k|        sum5_ptrs[3] = sum5_rows[0];
 1138|  8.03k|        sum5_ptrs[4] = sum5_rows[0];
 1139|       |
 1140|  8.03k|        sumsq3_ptrs[0] = sumsq3_rows[0];
 1141|  8.03k|        sumsq3_ptrs[1] = sumsq3_rows[0];
 1142|  8.03k|        sumsq3_ptrs[2] = sumsq3_rows[0];
 1143|  8.03k|        sum3_ptrs[0] = sum3_rows[0];
 1144|  8.03k|        sum3_ptrs[1] = sum3_rows[0];
 1145|  8.03k|        sum3_ptrs[2] = sum3_rows[0];
 1146|       |
 1147|  8.03k|        sgr_box35_row_h(sumsq3_rows[0], sum3_rows[0],
 1148|  8.03k|                        sumsq5_rows[0], sum5_rows[0],
 1149|  8.03k|                        left, src, w, edges);
 1150|  8.03k|        left++;
 1151|  8.03k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  8.03k|#define PXSTRIDE(x) (x)
  ------------------
 1152|       |
 1153|  8.03k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1154|  8.03k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  8.03k|#define BITDEPTH_MAX 0xff
  ------------------
 1155|  8.03k|        rotate(A3_ptrs, B3_ptrs, 4);
 1156|       |
 1157|  8.03k|        if (--h <= 0)
  ------------------
  |  Branch (1157:13): [True: 1.36k, False: 6.67k]
  ------------------
 1158|  1.36k|            goto vert_1;
 1159|       |
 1160|  6.67k|        sumsq5_ptrs[4] = sumsq5_rows[1];
 1161|  6.67k|        sum5_ptrs[4] = sum5_rows[1];
 1162|       |
 1163|  6.67k|        sumsq3_ptrs[2] = sumsq3_rows[1];
 1164|  6.67k|        sum3_ptrs[2] = sum3_rows[1];
 1165|       |
 1166|  6.67k|        sgr_box35_row_h(sumsq3_rows[1], sum3_rows[1],
 1167|  6.67k|                        sumsq5_rows[1], sum5_rows[1],
 1168|  6.67k|                        left, src, w, edges);
 1169|  6.67k|        left++;
 1170|  6.67k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  6.67k|#define PXSTRIDE(x) (x)
  ------------------
 1171|       |
 1172|  6.67k|        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1173|  6.67k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  6.67k|#define BITDEPTH_MAX 0xff
  ------------------
 1174|  6.67k|        rotate(A5_ptrs, B5_ptrs, 2);
 1175|  6.67k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1176|  6.67k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  6.67k|#define BITDEPTH_MAX 0xff
  ------------------
 1177|  6.67k|        rotate(A3_ptrs, B3_ptrs, 4);
 1178|       |
 1179|  6.67k|        if (--h <= 0)
  ------------------
  |  Branch (1179:13): [True: 193, False: 6.48k]
  ------------------
 1180|    193|            goto vert_2;
 1181|       |
 1182|  6.48k|        sumsq5_ptrs[3] = sumsq5_rows[2];
 1183|  6.48k|        sumsq5_ptrs[4] = sumsq5_rows[3];
 1184|  6.48k|        sum5_ptrs[3] = sum5_rows[2];
 1185|  6.48k|        sum5_ptrs[4] = sum5_rows[3];
 1186|       |
 1187|  6.48k|        sumsq3_ptrs[2] = sumsq3_rows[2];
 1188|  6.48k|        sum3_ptrs[2] = sum3_rows[2];
 1189|       |
 1190|  6.48k|        sgr_box35_row_h(sumsq3_rows[2], sum3_rows[2],
 1191|  6.48k|                        sumsq5_rows[2], sum5_rows[2],
 1192|  6.48k|                        left, src, w, edges);
 1193|  6.48k|        left++;
 1194|  6.48k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  6.48k|#define PXSTRIDE(x) (x)
  ------------------
 1195|       |
 1196|  6.48k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1197|  6.48k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  6.48k|#define BITDEPTH_MAX 0xff
  ------------------
 1198|  6.48k|        rotate(A3_ptrs, B3_ptrs, 4);
 1199|       |
 1200|  6.48k|        if (--h <= 0)
  ------------------
  |  Branch (1200:13): [True: 242, False: 6.23k]
  ------------------
 1201|    242|            goto odd;
 1202|       |
 1203|  6.23k|        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1204|  6.23k|                        sumsq5_rows[3], sum5_rows[3],
 1205|  6.23k|                        left, src, w, edges);
 1206|  6.23k|        left++;
 1207|  6.23k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|  6.23k|#define PXSTRIDE(x) (x)
  ------------------
 1208|       |
 1209|  6.23k|        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1210|  6.23k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  6.23k|#define BITDEPTH_MAX 0xff
  ------------------
 1211|  6.23k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1212|  6.23k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  6.23k|#define BITDEPTH_MAX 0xff
  ------------------
 1213|  6.23k|        sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1214|  6.23k|                       w, 2, params->sgr.w0, params->sgr.w1
 1215|  6.23k|                       HIGHBD_TAIL_SUFFIX);
 1216|       |
 1217|  6.23k|        if (--h <= 0)
  ------------------
  |  Branch (1217:13): [True: 89, False: 6.14k]
  ------------------
 1218|     89|            goto vert_2;
 1219|       |
 1220|       |        // ptrs are rotated by 2; both [3] and [4] now point at rows[0]; set
 1221|       |        // one of them to point at the previously unused rows[4].
 1222|  6.14k|        sumsq5_ptrs[3] = sumsq5_rows[4];
 1223|  6.14k|        sum5_ptrs[3] = sum5_rows[4];
 1224|  6.14k|    }
 1225|       |
 1226|   511k|    do {
 1227|   511k|        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1228|   511k|                        sumsq5_ptrs[3], sum5_ptrs[3],
 1229|   511k|                        left, src, w, edges);
 1230|   511k|        left++;
 1231|   511k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|   511k|#define PXSTRIDE(x) (x)
  ------------------
 1232|       |
 1233|   511k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1234|   511k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|   511k|#define BITDEPTH_MAX 0xff
  ------------------
 1235|   511k|        rotate(A3_ptrs, B3_ptrs, 4);
 1236|       |
 1237|   511k|        if (--h <= 0)
  ------------------
  |  Branch (1237:13): [True: 1.68k, False: 509k]
  ------------------
 1238|  1.68k|            goto odd;
 1239|       |
 1240|   509k|        sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1241|   509k|                        sumsq5_ptrs[4], sum5_ptrs[4],
 1242|   509k|                        left, src, w, edges);
 1243|   509k|        left++;
 1244|   509k|        src += PXSTRIDE(stride);
  ------------------
  |  |   53|   509k|#define PXSTRIDE(x) (x)
  ------------------
 1245|       |
 1246|   509k|        sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1247|   509k|                      w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|   509k|#define BITDEPTH_MAX 0xff
  ------------------
 1248|   509k|        sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1249|   509k|                      w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|   509k|#define BITDEPTH_MAX 0xff
  ------------------
 1250|   509k|        sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1251|   509k|                       w, 2, params->sgr.w0, params->sgr.w1
 1252|   509k|                       HIGHBD_TAIL_SUFFIX);
 1253|   509k|    } while (--h > 0);
  ------------------
  |  Branch (1253:14): [True: 492k, False: 16.8k]
  ------------------
 1254|       |
 1255|  16.8k|    if (!(edges & LR_HAVE_BOTTOM))
  ------------------
  |  Branch (1255:9): [True: 690, False: 16.1k]
  ------------------
 1256|    690|        goto vert_2;
 1257|       |
 1258|  16.1k|    sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1259|  16.1k|                    sumsq5_ptrs[3], sum5_ptrs[3],
 1260|  16.1k|                    NULL, lpf_bottom, w, edges);
 1261|  16.1k|    lpf_bottom += PXSTRIDE(stride);
  ------------------
  |  |   53|  16.1k|#define PXSTRIDE(x) (x)
  ------------------
 1262|  16.1k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1263|  16.1k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  16.1k|#define BITDEPTH_MAX 0xff
  ------------------
 1264|  16.1k|    rotate(A3_ptrs, B3_ptrs, 4);
 1265|       |
 1266|  16.1k|    sgr_box35_row_h(sumsq3_ptrs[2], sum3_ptrs[2],
 1267|  16.1k|                    sumsq5_ptrs[4], sum5_ptrs[4],
 1268|  16.1k|                    NULL, lpf_bottom, w, edges);
 1269|       |
 1270|  18.8k|output_2:
 1271|  18.8k|    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1272|  18.8k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  18.8k|#define BITDEPTH_MAX 0xff
  ------------------
 1273|  18.8k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1274|  18.8k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  18.8k|#define BITDEPTH_MAX 0xff
  ------------------
 1275|  18.8k|    sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1276|  18.8k|                   w, 2, params->sgr.w0, params->sgr.w1
 1277|  18.8k|                   HIGHBD_TAIL_SUFFIX);
 1278|  18.8k|    return;
 1279|       |
 1280|  2.65k|vert_2:
 1281|       |    // Duplicate the last row twice more
 1282|  2.65k|    sumsq5_ptrs[3] = sumsq5_ptrs[2];
 1283|  2.65k|    sumsq5_ptrs[4] = sumsq5_ptrs[2];
 1284|  2.65k|    sum5_ptrs[3] = sum5_ptrs[2];
 1285|  2.65k|    sum5_ptrs[4] = sum5_ptrs[2];
 1286|       |
 1287|  2.65k|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1288|  2.65k|    sum3_ptrs[2] = sum3_ptrs[1];
 1289|  2.65k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1290|  2.65k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  2.65k|#define BITDEPTH_MAX 0xff
  ------------------
 1291|  2.65k|    rotate(A3_ptrs, B3_ptrs, 4);
 1292|       |
 1293|  2.65k|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1294|  2.65k|    sum3_ptrs[2] = sum3_ptrs[1];
 1295|       |
 1296|  2.65k|    goto output_2;
 1297|       |
 1298|  1.92k|odd:
 1299|       |    // Copy the last row as padding once
 1300|  1.92k|    sumsq5_ptrs[4] = sumsq5_ptrs[3];
 1301|  1.92k|    sum5_ptrs[4] = sum5_ptrs[3];
 1302|       |
 1303|  1.92k|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1304|  1.92k|    sum3_ptrs[2] = sum3_ptrs[1];
 1305|       |
 1306|  1.92k|    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1307|  1.92k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.92k|#define BITDEPTH_MAX 0xff
  ------------------
 1308|  1.92k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1309|  1.92k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  1.92k|#define BITDEPTH_MAX 0xff
  ------------------
 1310|  1.92k|    sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1311|  1.92k|                   w, 2, params->sgr.w0, params->sgr.w1
 1312|  1.92k|                   HIGHBD_TAIL_SUFFIX);
 1313|       |
 1314|  5.37k|output_1:
 1315|       |    // Duplicate the last row twice more
 1316|  5.37k|    sumsq5_ptrs[3] = sumsq5_ptrs[2];
 1317|  5.37k|    sumsq5_ptrs[4] = sumsq5_ptrs[2];
 1318|  5.37k|    sum5_ptrs[3] = sum5_ptrs[2];
 1319|  5.37k|    sum5_ptrs[4] = sum5_ptrs[2];
 1320|       |
 1321|  5.37k|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1322|  5.37k|    sum3_ptrs[2] = sum3_ptrs[1];
 1323|       |
 1324|  5.37k|    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1325|  5.37k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  5.37k|#define BITDEPTH_MAX 0xff
  ------------------
 1326|  5.37k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1327|  5.37k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  5.37k|#define BITDEPTH_MAX 0xff
  ------------------
 1328|  5.37k|    rotate(A3_ptrs, B3_ptrs, 4);
 1329|       |    // Output only one row
 1330|  5.37k|    sgr_finish_mix(&dst, stride, A5_ptrs, B5_ptrs, A3_ptrs, B3_ptrs,
 1331|  5.37k|                   w, 1, params->sgr.w0, params->sgr.w1
 1332|  5.37k|                   HIGHBD_TAIL_SUFFIX);
 1333|  5.37k|    return;
 1334|       |
 1335|  3.45k|vert_1:
 1336|       |    // Copy the last row as padding once
 1337|  3.45k|    sumsq5_ptrs[4] = sumsq5_ptrs[3];
 1338|  3.45k|    sum5_ptrs[4] = sum5_ptrs[3];
 1339|       |
 1340|  3.45k|    sumsq3_ptrs[2] = sumsq3_ptrs[1];
 1341|  3.45k|    sum3_ptrs[2] = sum3_ptrs[1];
 1342|       |
 1343|  3.45k|    sgr_box5_vert(sumsq5_ptrs, sum5_ptrs, A5_ptrs[1], B5_ptrs[1],
 1344|  3.45k|                  w, params->sgr.s0, BITDEPTH_MAX);
  ------------------
  |  |   59|  3.45k|#define BITDEPTH_MAX 0xff
  ------------------
 1345|  3.45k|    rotate(A5_ptrs, B5_ptrs, 2);
 1346|  3.45k|    sgr_box3_vert(sumsq3_ptrs, sum3_ptrs, A3_ptrs[3], B3_ptrs[3],
 1347|  3.45k|                  w, params->sgr.s1, BITDEPTH_MAX);
  ------------------
  |  |   59|  3.45k|#define BITDEPTH_MAX 0xff
  ------------------
 1348|  3.45k|    rotate(A3_ptrs, B3_ptrs, 4);
 1349|       |
 1350|  3.45k|    goto output_1;
 1351|  1.92k|}
looprestoration_tmpl.c:sgr_box35_row_h:
  464|  1.13M|{
  465|  1.13M|    sgr_box3_row_h(sumsq3, sum3, left, src, w, edges);
  466|  1.13M|    sgr_box5_row_h(sumsq5, sum5, left, src, w, edges);
  467|  1.13M|}
looprestoration_tmpl.c:sgr_finish_mix:
  663|   540k|{
  664|   540k|    ALIGN_STK_16(coef, tmp5, 2*FILTER_OUT_STRIDE,);
  ------------------
  |  |  100|   540k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|   540k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  665|   540k|    ALIGN_STK_16(coef, tmp3, 2*FILTER_OUT_STRIDE,);
  ------------------
  |  |  100|   540k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|   540k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  666|       |
  667|   540k|    sgr_finish_filter2(tmp5, *dst, stride, A5_ptrs, B5_ptrs, w, h);
  668|   540k|    sgr_finish_filter_row1(tmp3, *dst, A3_ptrs, B3_ptrs, w);
  669|   540k|    if (h > 1)
  ------------------
  |  Branch (669:9): [True: 536k, False: 4.69k]
  ------------------
  670|   536k|        sgr_finish_filter_row1(tmp3 + FILTER_OUT_STRIDE, *dst + PXSTRIDE(stride),
  ------------------
  |  |  572|   536k|#define FILTER_OUT_STRIDE (384)
  ------------------
                      sgr_finish_filter_row1(tmp3 + FILTER_OUT_STRIDE, *dst + PXSTRIDE(stride),
  ------------------
  |  |   53|   536k|#define PXSTRIDE(x) (x)
  ------------------
  671|   536k|                               &A3_ptrs[1], &B3_ptrs[1], w);
  672|   540k|    sgr_weighted2(*dst, stride, tmp5, tmp3, w, h, w0, w1 HIGHBD_TAIL_SUFFIX);
  673|   540k|    *dst += h*PXSTRIDE(stride);
  ------------------
  |  |   53|   540k|#define PXSTRIDE(x) (x)
  ------------------
  674|   540k|    rotate(A5_ptrs, B5_ptrs, 2);
  675|   540k|    rotate(A3_ptrs, B3_ptrs, 4);
  676|   540k|}
looprestoration_tmpl.c:sgr_weighted2:
  616|   540k|{
  617|  1.61M|    for (int j = 0; j < h; j++) {
  ------------------
  |  Branch (617:21): [True: 1.07M, False: 540k]
  ------------------
  618|  96.6M|        for (int i = 0; i < w; i++) {
  ------------------
  |  Branch (618:25): [True: 95.6M, False: 1.07M]
  ------------------
  619|  95.6M|            const int v = w0 * t1[i] + w1 * t2[i];
  620|  95.6M|            dst[i] = iclip_pixel(dst[i] + ((v + (1 << 10)) >> 11));
  ------------------
  |  |   49|  95.6M|#define iclip_pixel iclip_u8
  ------------------
  621|  95.6M|        }
  622|  1.07M|        dst += PXSTRIDE(dst_stride);
  ------------------
  |  |   53|  1.07M|#define PXSTRIDE(x) (x)
  ------------------
  623|  1.07M|        t1 += FILTER_OUT_STRIDE;
  ------------------
  |  |  572|  1.07M|#define FILTER_OUT_STRIDE (384)
  ------------------
  624|  1.07M|        t2 += FILTER_OUT_STRIDE;
  ------------------
  |  |  572|  1.07M|#define FILTER_OUT_STRIDE (384)
  ------------------
  625|  1.07M|    }
  626|   540k|}
dav1d_loop_restoration_dsp_init_16bpc:
 1367|  5.11k|{
 1368|  5.11k|    c->wiener[0] = c->wiener[1] = wiener_c;
 1369|  5.11k|    c->sgr[0] = sgr_5x5_c;
 1370|  5.11k|    c->sgr[1] = sgr_3x3_c;
 1371|  5.11k|    c->sgr[2] = sgr_mix_c;
 1372|       |
 1373|  5.11k|#if HAVE_ASM
 1374|       |#if ARCH_AARCH64 || ARCH_ARM
 1375|       |    loop_restoration_dsp_init_arm(c, bpc);
 1376|       |#elif ARCH_LOONGARCH64
 1377|       |    loop_restoration_dsp_init_loongarch(c, bpc);
 1378|       |#elif ARCH_PPC64LE
 1379|       |    loop_restoration_dsp_init_ppc(c, bpc);
 1380|       |#elif ARCH_X86
 1381|       |    loop_restoration_dsp_init_x86(c, bpc);
 1382|  5.11k|#endif
 1383|  5.11k|#endif
 1384|  5.11k|}

dav1d_lr_sbrow_8bpc:
  171|   175k|{
  172|   175k|    const int offset_y = 8 * !!sby;
  173|   175k|    const ptrdiff_t *const dst_stride = f->sr_cur.p.stride;
  174|   175k|    const int restore_planes = f->lf.restore_planes;
  175|   175k|    const int sb128 = f->seq_hdr->sb128;
  176|   175k|    const int not_last = sby + 1 < f->sbh;
  177|       |
  178|   175k|    if (restore_planes & LR_RESTORE_Y) {
  ------------------
  |  Branch (178:9): [True: 171k, False: 3.71k]
  ------------------
  179|   171k|        const int h = f->sr_cur.p.p.h;
  180|   171k|        const int w = f->sr_cur.p.p.w;
  181|   171k|        const int next_row_y = (sby + 1) << (6 + sb128);
  182|   171k|        const int row_h = imin(next_row_y - 8 * not_last, h);
  183|   171k|        const int y_stripe = (sby << (6 + sb128)) - offset_y;
  184|   171k|        lr_sbrow(f, dst[0] - offset_y * PXSTRIDE(dst_stride[0]), y_stripe, w,
  ------------------
  |  |   53|   171k|#define PXSTRIDE(x) (x)
  ------------------
  185|   171k|                 h, row_h, 0);
  186|   171k|    }
  187|   175k|    if (restore_planes & (LR_RESTORE_U | LR_RESTORE_V)) {
  ------------------
  |  Branch (187:9): [True: 8.27k, False: 167k]
  ------------------
  188|  8.27k|        const int ss_ver = f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  189|  8.27k|        const int ss_hor = f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  190|  8.27k|        const int h = (f->sr_cur.p.p.h + ss_ver) >> ss_ver;
  191|  8.27k|        const int w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
  192|  8.27k|        const int next_row_y = (sby + 1) << ((6 - ss_ver) + sb128);
  193|  8.27k|        const int row_h = imin(next_row_y - (8 >> ss_ver) * not_last, h);
  194|  8.27k|        const int offset_uv = offset_y >> ss_ver;
  195|  8.27k|        const int y_stripe = (sby << ((6 - ss_ver) + sb128)) - offset_uv;
  196|  8.27k|        if (restore_planes & LR_RESTORE_U)
  ------------------
  |  Branch (196:13): [True: 6.17k, False: 2.10k]
  ------------------
  197|  6.17k|            lr_sbrow(f, dst[1] - offset_uv * PXSTRIDE(dst_stride[1]), y_stripe,
  ------------------
  |  |   53|  6.17k|#define PXSTRIDE(x) (x)
  ------------------
  198|  6.17k|                     w, h, row_h, 1);
  199|       |
  200|  8.27k|        if (restore_planes & LR_RESTORE_V)
  ------------------
  |  Branch (200:13): [True: 6.17k, False: 2.10k]
  ------------------
  201|  6.17k|            lr_sbrow(f, dst[2] - offset_uv * PXSTRIDE(dst_stride[1]), y_stripe,
  ------------------
  |  |   53|  6.17k|#define PXSTRIDE(x) (x)
  ------------------
  202|  6.17k|                     w, h, row_h, 2);
  203|  8.27k|    }
  204|   175k|}
lr_apply_tmpl.c:lr_sbrow:
  109|   239k|{
  110|   239k|    const int chroma = !!plane;
  111|   239k|    const int ss_ver = chroma & (f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420);
  112|   239k|    const int ss_hor = chroma & (f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444);
  113|   239k|    const ptrdiff_t p_stride = f->sr_cur.p.stride[chroma];
  114|       |
  115|   239k|    const int unit_size_log2 = f->frame_hdr->restoration.unit_size[!!plane];
  116|   239k|    const int unit_size = 1 << unit_size_log2;
  117|   239k|    const int half_unit_size = unit_size >> 1;
  118|   239k|    const int max_unit_size = unit_size + half_unit_size;
  119|       |
  120|       |    // Y coordinate of the sbrow (y is 8 luma pixel rows above row_y)
  121|   239k|    const int row_y = y + ((8 >> ss_ver) * !!y);
  122|       |
  123|       |    // FIXME This is an ugly hack to lookup the proper AV1Filter unit for
  124|       |    // chroma planes. Question: For Multithreaded decoding, is it better
  125|       |    // to store the chroma LR information with collocated Luma information?
  126|       |    // In other words. For a chroma restoration unit locate at 128,128 and
  127|       |    // with a 4:2:0 chroma subsampling, do we store the filter information at
  128|       |    // the AV1Filter unit located at (128,128) or (256,256)
  129|       |    // TODO Support chroma subsampling.
  130|   239k|    const int shift_hor = 7 - ss_hor;
  131|       |
  132|       |    /* maximum sbrow height is 128 + 8 rows offset */
  133|   239k|    ALIGN_STK_16(pixel, pre_lr_border, 2, [128 + 8][4]);
  ------------------
  |  |  100|   239k|    ALIGN(type var[sz1d]sznd, ALIGN_16_VAL)
  |  |  ------------------
  |  |  |  |   86|   239k|    line __attribute__((aligned(align)))
  |  |  ------------------
  ------------------
  134|   239k|    const Av1RestorationUnit *lr[2];
  135|       |
  136|   239k|    enum LrEdgeFlags edges = (y > 0 ? LR_HAVE_TOP : 0) | LR_HAVE_RIGHT;
  ------------------
  |  Branch (136:31): [True: 220k, False: 18.9k]
  ------------------
  137|       |
  138|   239k|    int aligned_unit_pos = row_y & ~(unit_size - 1);
  139|   239k|    if (aligned_unit_pos && aligned_unit_pos + half_unit_size > h)
  ------------------
  |  Branch (139:9): [True: 214k, False: 25.6k]
  |  Branch (139:29): [True: 4.76k, False: 209k]
  ------------------
  140|  4.76k|        aligned_unit_pos -= unit_size;
  141|   239k|    aligned_unit_pos <<= ss_ver;
  142|   239k|    const int sb_idx = (aligned_unit_pos >> 7) * f->sr_sb128w;
  143|   239k|    const int unit_idx = ((aligned_unit_pos >> 6) & 1) << 1;
  144|   239k|    lr[0] = &f->lf.lr_mask[sb_idx].lr[plane][unit_idx];
  145|   239k|    int restore = lr[0]->type != DAV1D_RESTORATION_NONE;
  146|   239k|    int x = 0, bit = 0;
  147|   239k|    const int backup_h = row_h - y;
  148|   289k|    for (; x + max_unit_size <= w; p += unit_size, edges |= LR_HAVE_LEFT, bit ^= 1) {
  ------------------
  |  Branch (148:12): [True: 49.9k, False: 239k]
  ------------------
  149|  49.9k|        const int next_x = x + unit_size;
  150|  49.9k|        const int next_u_idx = unit_idx + ((next_x >> (shift_hor - 1)) & 1);
  151|  49.9k|        lr[!bit] =
  152|  49.9k|            &f->lf.lr_mask[sb_idx + (next_x >> shift_hor)].lr[plane][next_u_idx];
  153|  49.9k|        const int restore_next = lr[!bit]->type != DAV1D_RESTORATION_NONE;
  154|  49.9k|        if (restore_next)
  ------------------
  |  Branch (154:13): [True: 26.5k, False: 23.4k]
  ------------------
  155|  26.5k|            backup4xU(pre_lr_border[bit], p + unit_size - 4, p_stride, backup_h);
  156|  49.9k|        if (restore)
  ------------------
  |  Branch (156:13): [True: 26.6k, False: 23.2k]
  ------------------
  157|  26.6k|            lr_stripe(f, p, pre_lr_border[!bit], x, y, plane, unit_size, row_h,
  158|  26.6k|                      lr[bit], edges);
  159|  49.9k|        x = next_x;
  160|  49.9k|        restore = restore_next;
  161|  49.9k|    }
  162|   239k|    if (restore) {
  ------------------
  |  Branch (162:9): [True: 25.2k, False: 214k]
  ------------------
  163|  25.2k|        edges &= ~LR_HAVE_RIGHT;
  164|  25.2k|        const int unit_w = w - x;
  165|  25.2k|        lr_stripe(f, p, pre_lr_border[!bit], x, y, plane, unit_w, row_h, lr[bit], edges);
  166|  25.2k|    }
  167|   239k|}
lr_apply_tmpl.c:backup4xU:
  102|  26.5k|{
  103|  2.49M|    for (; u > 0; u--, dst++, src += PXSTRIDE(src_stride))
  ------------------
  |  |   53|  2.47M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (103:12): [True: 2.47M, False: 26.5k]
  ------------------
  104|  2.47M|        pixel_copy(dst, src, 4);
  ------------------
  |  |   47|  2.47M|#define pixel_copy memcpy
  ------------------
  105|  26.5k|}
lr_apply_tmpl.c:lr_stripe:
   40|  51.9k|{
   41|  51.9k|    const Dav1dDSPContext *const dsp = f->dsp;
   42|  51.9k|    const int chroma = !!plane;
   43|  51.9k|    const int ss_ver = chroma & (f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420);
   44|  51.9k|    const ptrdiff_t stride = f->sr_cur.p.stride[chroma];
   45|  51.9k|    const int sby = (y + (y ? 8 << ss_ver : 0)) >> (6 - ss_ver + f->seq_hdr->sb128);
  ------------------
  |  Branch (45:27): [True: 33.6k, False: 18.2k]
  ------------------
   46|  51.9k|    const int have_tt = f->c->n_tc > 1;
   47|  51.9k|    const pixel *lpf = f->lf.lr_lpf_line[plane] +
   48|  51.9k|        have_tt * (sby * (4 << f->seq_hdr->sb128) - 4) * PXSTRIDE(stride) + x;
  ------------------
  |  |   53|  51.9k|#define PXSTRIDE(x) (x)
  ------------------
   49|       |
   50|       |    // The first stripe of the frame is shorter by 8 luma pixel rows.
   51|  51.9k|    int stripe_h = imin((64 - 8 * !y) >> ss_ver, row_h - y);
   52|       |
   53|  51.9k|    looprestorationfilter_fn lr_fn;
   54|  51.9k|    LooprestorationParams params;
   55|  51.9k|    if (lr->type == DAV1D_RESTORATION_WIENER) {
  ------------------
  |  Branch (55:9): [True: 21.7k, False: 30.1k]
  ------------------
   56|  21.7k|        int16_t (*const filter)[8] = params.filter;
   57|  21.7k|        filter[0][0] = filter[0][6] = lr->filter_h[0];
   58|  21.7k|        filter[0][1] = filter[0][5] = lr->filter_h[1];
   59|  21.7k|        filter[0][2] = filter[0][4] = lr->filter_h[2];
   60|  21.7k|        filter[0][3] = -(filter[0][0] + filter[0][1] + filter[0][2]) * 2;
   61|       |#if BITDEPTH != 8
   62|       |        /* For 8-bit SIMD it's beneficial to handle the +128 separately
   63|       |         * in order to avoid overflows. */
   64|       |        filter[0][3] += 128;
   65|       |#endif
   66|       |
   67|  21.7k|        filter[1][0] = filter[1][6] = lr->filter_v[0];
   68|  21.7k|        filter[1][1] = filter[1][5] = lr->filter_v[1];
   69|  21.7k|        filter[1][2] = filter[1][4] = lr->filter_v[2];
   70|  21.7k|        filter[1][3] = 128 - (filter[1][0] + filter[1][1] + filter[1][2]) * 2;
   71|       |
   72|  21.7k|        lr_fn = dsp->lr.wiener[!(filter[0][0] | filter[1][0])];
   73|  30.1k|    } else {
   74|  30.1k|        assert(lr->type >= DAV1D_RESTORATION_SGRPROJ);
  ------------------
  |  Branch (74:9): [True: 30.1k, False: 18.4E]
  ------------------
   75|  30.1k|        const int sgr_idx = lr->type - DAV1D_RESTORATION_SGRPROJ;
   76|  30.1k|        const uint16_t *const sgr_params = dav1d_sgr_params[sgr_idx];
   77|  30.1k|        params.sgr.s0 = sgr_params[0];
   78|  30.1k|        params.sgr.s1 = sgr_params[1];
   79|  30.1k|        params.sgr.w0 = lr->sgr_weights[0];
   80|  30.1k|        params.sgr.w1 = 128 - (lr->sgr_weights[0] + lr->sgr_weights[1]);
   81|       |
   82|  30.1k|        lr_fn = dsp->lr.sgr[!!sgr_params[0] + !!sgr_params[1] * 2 - 1];
   83|  30.1k|    }
   84|       |
   85|  92.8k|    while (y + stripe_h <= row_h) {
  ------------------
  |  Branch (85:12): [True: 92.8k, False: 0]
  ------------------
   86|       |        // Change the HAVE_BOTTOM bit in edges to (sby + 1 != f->sbh || y + stripe_h != row_h)
   87|  92.8k|        edges ^= (-(sby + 1 != f->sbh || y + stripe_h != row_h) ^ edges) & LR_HAVE_BOTTOM;
  ------------------
  |  Branch (87:21): [True: 55.5k, False: 37.3k]
  |  Branch (87:42): [True: 19.1k, False: 18.1k]
  ------------------
   88|  92.8k|        lr_fn(p, stride, left, lpf, unit_w, stripe_h, &params, edges HIGHBD_CALL_SUFFIX);
   89|       |
   90|  92.8k|        left += stripe_h;
   91|  92.8k|        y += stripe_h;
   92|  92.8k|        p += stripe_h * PXSTRIDE(stride);
  ------------------
  |  |   53|  92.8k|#define PXSTRIDE(x) (x)
  ------------------
   93|  92.8k|        edges |= LR_HAVE_TOP;
   94|  92.8k|        stripe_h = imin(64 >> ss_ver, row_h - y);
   95|  92.8k|        if (stripe_h == 0) break;
  ------------------
  |  Branch (95:13): [True: 51.9k, False: 40.8k]
  ------------------
   96|  40.8k|        lpf += 4 * PXSTRIDE(stride);
  ------------------
  |  |   53|  40.8k|#define PXSTRIDE(x) (x)
  ------------------
   97|  40.8k|    }
   98|  51.9k|}
dav1d_lr_sbrow_16bpc:
  171|  42.4k|{
  172|  42.4k|    const int offset_y = 8 * !!sby;
  173|  42.4k|    const ptrdiff_t *const dst_stride = f->sr_cur.p.stride;
  174|  42.4k|    const int restore_planes = f->lf.restore_planes;
  175|  42.4k|    const int sb128 = f->seq_hdr->sb128;
  176|  42.4k|    const int not_last = sby + 1 < f->sbh;
  177|       |
  178|  42.4k|    if (restore_planes & LR_RESTORE_Y) {
  ------------------
  |  Branch (178:9): [True: 41.4k, False: 991]
  ------------------
  179|  41.4k|        const int h = f->sr_cur.p.p.h;
  180|  41.4k|        const int w = f->sr_cur.p.p.w;
  181|  41.4k|        const int next_row_y = (sby + 1) << (6 + sb128);
  182|  41.4k|        const int row_h = imin(next_row_y - 8 * not_last, h);
  183|  41.4k|        const int y_stripe = (sby << (6 + sb128)) - offset_y;
  184|  41.4k|        lr_sbrow(f, dst[0] - offset_y * PXSTRIDE(dst_stride[0]), y_stripe, w,
  185|  41.4k|                 h, row_h, 0);
  186|  41.4k|    }
  187|  42.4k|    if (restore_planes & (LR_RESTORE_U | LR_RESTORE_V)) {
  ------------------
  |  Branch (187:9): [True: 8.85k, False: 33.5k]
  ------------------
  188|  8.85k|        const int ss_ver = f->sr_cur.p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  189|  8.85k|        const int ss_hor = f->sr_cur.p.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  190|  8.85k|        const int h = (f->sr_cur.p.p.h + ss_ver) >> ss_ver;
  191|  8.85k|        const int w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
  192|  8.85k|        const int next_row_y = (sby + 1) << ((6 - ss_ver) + sb128);
  193|  8.85k|        const int row_h = imin(next_row_y - (8 >> ss_ver) * not_last, h);
  194|  8.85k|        const int offset_uv = offset_y >> ss_ver;
  195|  8.85k|        const int y_stripe = (sby << ((6 - ss_ver) + sb128)) - offset_uv;
  196|  8.85k|        if (restore_planes & LR_RESTORE_U)
  ------------------
  |  Branch (196:13): [True: 6.56k, False: 2.28k]
  ------------------
  197|  6.56k|            lr_sbrow(f, dst[1] - offset_uv * PXSTRIDE(dst_stride[1]), y_stripe,
  198|  6.56k|                     w, h, row_h, 1);
  199|       |
  200|  8.85k|        if (restore_planes & LR_RESTORE_V)
  ------------------
  |  Branch (200:13): [True: 7.81k, False: 1.03k]
  ------------------
  201|  7.81k|            lr_sbrow(f, dst[2] - offset_uv * PXSTRIDE(dst_stride[1]), y_stripe,
  202|  7.81k|                     w, h, row_h, 2);
  203|  8.85k|    }
  204|  42.4k|}

dav1d_mc_dsp_init_8bpc:
  960|  3.46k|COLD void bitfn(dav1d_mc_dsp_init)(Dav1dMCDSPContext *const c) {
  961|  3.46k|#define init_mc_fns(type, name) do { \
  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  966|  3.46k|} while (0)
  967|       |
  968|  3.46k|    init_mc_fns(FILTER_2D_8TAP_REGULAR,        8tap_regular);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  969|  3.46k|    init_mc_fns(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  970|  3.46k|    init_mc_fns(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  971|  3.46k|    init_mc_fns(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  972|  3.46k|    init_mc_fns(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  973|  3.46k|    init_mc_fns(FILTER_2D_8TAP_SHARP,          8tap_sharp);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  974|  3.46k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  975|  3.46k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH,         8tap_smooth);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  976|  3.46k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  977|  3.46k|    init_mc_fns(FILTER_2D_BILINEAR,            bilin);
  ------------------
  |  |  961|  3.46k|#define init_mc_fns(type, name) do { \
  |  |  962|  3.46k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  3.46k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  3.46k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  3.46k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  3.46k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 3.46k]
  |  |  ------------------
  ------------------
  978|       |
  979|  3.46k|    c->avg      = avg_c;
  980|  3.46k|    c->w_avg    = w_avg_c;
  981|  3.46k|    c->mask     = mask_c;
  982|  3.46k|    c->blend    = blend_c;
  983|  3.46k|    c->blend_v  = blend_v_c;
  984|  3.46k|    c->blend_h  = blend_h_c;
  985|  3.46k|    c->w_mask[0] = w_mask_444_c;
  986|  3.46k|    c->w_mask[1] = w_mask_422_c;
  987|  3.46k|    c->w_mask[2] = w_mask_420_c;
  988|  3.46k|    c->warp8x8  = warp_affine_8x8_c;
  989|  3.46k|    c->warp8x8t = warp_affine_8x8t_c;
  990|  3.46k|    c->emu_edge = emu_edge_c;
  991|  3.46k|    c->resize   = resize_c;
  992|       |
  993|  3.46k|#if HAVE_ASM
  994|       |#if ARCH_AARCH64 || ARCH_ARM
  995|       |    mc_dsp_init_arm(c);
  996|       |#elif ARCH_LOONGARCH64
  997|       |    mc_dsp_init_loongarch(c);
  998|       |#elif ARCH_PPC64LE
  999|       |    mc_dsp_init_ppc(c);
 1000|       |#elif ARCH_RISCV
 1001|       |    mc_dsp_init_riscv(c);
 1002|       |#elif ARCH_X86
 1003|       |    mc_dsp_init_x86(c);
 1004|  3.46k|#endif
 1005|  3.46k|#endif
 1006|  3.46k|}
dav1d_mc_dsp_init_16bpc:
  960|  5.11k|COLD void bitfn(dav1d_mc_dsp_init)(Dav1dMCDSPContext *const c) {
  961|  5.11k|#define init_mc_fns(type, name) do { \
  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  966|  5.11k|} while (0)
  967|       |
  968|  5.11k|    init_mc_fns(FILTER_2D_8TAP_REGULAR,        8tap_regular);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  969|  5.11k|    init_mc_fns(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  970|  5.11k|    init_mc_fns(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  971|  5.11k|    init_mc_fns(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  972|  5.11k|    init_mc_fns(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  973|  5.11k|    init_mc_fns(FILTER_2D_8TAP_SHARP,          8tap_sharp);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  974|  5.11k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  975|  5.11k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH,         8tap_smooth);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  976|  5.11k|    init_mc_fns(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  977|  5.11k|    init_mc_fns(FILTER_2D_BILINEAR,            bilin);
  ------------------
  |  |  961|  5.11k|#define init_mc_fns(type, name) do { \
  |  |  962|  5.11k|    c->mc        [type] = put_##name##_c; \
  |  |  963|  5.11k|    c->mc_scaled [type] = put_##name##_scaled_c; \
  |  |  964|  5.11k|    c->mct       [type] = prep_##name##_c; \
  |  |  965|  5.11k|    c->mct_scaled[type] = prep_##name##_scaled_c; \
  |  |  966|  5.11k|} while (0)
  |  |  ------------------
  |  |  |  Branch (966:10): [Folded, False: 5.11k]
  |  |  ------------------
  ------------------
  978|       |
  979|  5.11k|    c->avg      = avg_c;
  980|  5.11k|    c->w_avg    = w_avg_c;
  981|  5.11k|    c->mask     = mask_c;
  982|  5.11k|    c->blend    = blend_c;
  983|  5.11k|    c->blend_v  = blend_v_c;
  984|  5.11k|    c->blend_h  = blend_h_c;
  985|  5.11k|    c->w_mask[0] = w_mask_444_c;
  986|  5.11k|    c->w_mask[1] = w_mask_422_c;
  987|  5.11k|    c->w_mask[2] = w_mask_420_c;
  988|  5.11k|    c->warp8x8  = warp_affine_8x8_c;
  989|  5.11k|    c->warp8x8t = warp_affine_8x8t_c;
  990|  5.11k|    c->emu_edge = emu_edge_c;
  991|  5.11k|    c->resize   = resize_c;
  992|       |
  993|  5.11k|#if HAVE_ASM
  994|       |#if ARCH_AARCH64 || ARCH_ARM
  995|       |    mc_dsp_init_arm(c);
  996|       |#elif ARCH_LOONGARCH64
  997|       |    mc_dsp_init_loongarch(c);
  998|       |#elif ARCH_PPC64LE
  999|       |    mc_dsp_init_ppc(c);
 1000|       |#elif ARCH_RISCV
 1001|       |    mc_dsp_init_riscv(c);
 1002|       |#elif ARCH_X86
 1003|       |    mc_dsp_init_x86(c);
 1004|  5.11k|#endif
 1005|  5.11k|#endif
 1006|  5.11k|}

dav1d_mem_pool_push:
  224|  1.60M|void dav1d_mem_pool_push(Dav1dMemPool *const pool, void *const ptr) {
  225|  1.60M|    pthread_mutex_lock(&pool->lock);
  226|  1.60M|    Dav1dMemPoolBuffer *const buf = (Dav1dMemPoolBuffer*)((uintptr_t)ptr - 64);
  227|  1.60M|    const int ref_cnt = --pool->ref_cnt;
  228|  1.60M|    if (!pool->end) {
  ------------------
  |  Branch (228:9): [True: 1.57M, False: 20.5k]
  ------------------
  229|  1.57M|        buf->next = pool->buf;
  230|  1.57M|        pool->buf = buf;
  231|  1.57M|        pthread_mutex_unlock(&pool->lock);
  232|  1.57M|        assert(ref_cnt > 0);
  ------------------
  |  Branch (232:9): [True: 1.57M, False: 18.4E]
  ------------------
  233|  1.57M|    } else {
  234|  20.5k|        pthread_mutex_unlock(&pool->lock);
  235|  20.5k|        dav1d_free_aligned(buf);
  ------------------
  |  |  136|  20.5k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  236|  20.6k|        if (!ref_cnt) mem_pool_destroy(pool);
  ------------------
  |  Branch (236:13): [True: 20.6k, False: 18.4E]
  ------------------
  237|  20.5k|    }
  238|  1.60M|}
dav1d_mem_pool_pop:
  240|  1.60M|void *dav1d_mem_pool_pop(Dav1dMemPool *const pool, const size_t size) {
  241|  1.60M|    pthread_mutex_lock(&pool->lock);
  242|  1.60M|    Dav1dMemPoolBuffer *buf = pool->buf;
  243|  1.60M|    pool->ref_cnt++;
  244|       |
  245|  1.60M|    if (buf) {
  ------------------
  |  Branch (245:9): [True: 1.50M, False: 97.6k]
  ------------------
  246|  1.50M|        pool->buf = buf->next;
  247|  1.50M|        pthread_mutex_unlock(&pool->lock);
  248|  1.50M|        if (buf->size != size) {
  ------------------
  |  Branch (248:13): [True: 20.7k, False: 1.48M]
  ------------------
  249|       |            /* Reallocate if the size has changed */
  250|  20.7k|            dav1d_free_aligned(buf);
  ------------------
  |  |  136|  20.7k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  251|  20.7k|            goto alloc;
  252|  20.7k|        }
  253|       |#if TRACK_HEAP_ALLOCATIONS
  254|       |        dav1d_track_reuse(pool->type);
  255|       |#endif
  256|  1.50M|    } else {
  257|  97.6k|        pthread_mutex_unlock(&pool->lock);
  258|   118k|alloc:
  259|   118k|        buf = dav1d_alloc_aligned(pool->type, size + 64, 64);
  ------------------
  |  |  134|   118k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  260|   118k|        if (!buf) {
  ------------------
  |  Branch (260:13): [True: 0, False: 118k]
  ------------------
  261|      0|            pthread_mutex_lock(&pool->lock);
  262|      0|            const int ref_cnt = --pool->ref_cnt;
  263|      0|            pthread_mutex_unlock(&pool->lock);
  264|      0|            if (!ref_cnt) mem_pool_destroy(pool);
  ------------------
  |  Branch (264:17): [True: 0, False: 0]
  ------------------
  265|      0|            return NULL;
  266|      0|        }
  267|   118k|        buf->size = size;
  268|   118k|    }
  269|       |
  270|  1.60M|    return (void*)((uintptr_t)buf + 64);
  271|  1.60M|}
dav1d_mem_pool_init:
  275|  65.9k|{
  276|  65.9k|    Dav1dMemPool *const pool = dav1d_malloc(ALLOC_COMMON_CTX,
  ------------------
  |  |  132|  65.9k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
  277|  65.9k|                                            sizeof(Dav1dMemPool));
  278|  65.9k|    if (pool) {
  ------------------
  |  Branch (278:9): [True: 65.9k, False: 0]
  ------------------
  279|  65.9k|        if (!pthread_mutex_init(&pool->lock, NULL)) {
  ------------------
  |  Branch (279:13): [True: 65.9k, False: 0]
  ------------------
  280|  65.9k|            pool->buf = NULL;
  281|  65.9k|            pool->ref_cnt = 1;
  282|  65.9k|            pool->end = 0;
  283|       |#if TRACK_HEAP_ALLOCATIONS
  284|       |            pool->type = type;
  285|       |#endif
  286|  65.9k|            *ppool = pool;
  287|  65.9k|            return 0;
  288|  65.9k|        }
  289|      0|        dav1d_free(pool);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
  290|      0|    }
  291|      0|    *ppool = NULL;
  292|      0|    return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  293|  65.9k|}
dav1d_mem_pool_end:
  295|  65.9k|COLD void dav1d_mem_pool_end(Dav1dMemPool *const pool) {
  296|  65.9k|    if (pool) {
  ------------------
  |  Branch (296:9): [True: 65.9k, False: 0]
  ------------------
  297|  65.9k|        pthread_mutex_lock(&pool->lock);
  298|  65.9k|        Dav1dMemPoolBuffer *buf = pool->buf;
  299|  65.9k|        const int ref_cnt = --pool->ref_cnt;
  300|  65.9k|        pool->buf = NULL;
  301|  65.9k|        pool->end = 1;
  302|  65.9k|        pthread_mutex_unlock(&pool->lock);
  303|       |
  304|   142k|        while (buf) {
  ------------------
  |  Branch (304:16): [True: 76.9k, False: 65.9k]
  ------------------
  305|  76.9k|            void *const ptr = buf;
  306|  76.9k|            buf = buf->next;
  307|  76.9k|            dav1d_free_aligned(ptr);
  ------------------
  |  |  136|  76.9k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  308|  76.9k|        }
  309|  65.9k|        if (!ref_cnt) mem_pool_destroy(pool);
  ------------------
  |  Branch (309:13): [True: 45.2k, False: 20.6k]
  ------------------
  310|  65.9k|    }
  311|  65.9k|}
mem.c:mem_pool_destroy:
  219|  65.9k|static COLD void mem_pool_destroy(Dav1dMemPool *const pool) {
  220|  65.9k|    pthread_mutex_destroy(&pool->lock);
  221|  65.9k|    dav1d_free(pool);
  ------------------
  |  |  135|  65.9k|#define dav1d_free(ptr) free(ptr)
  ------------------
  222|  65.9k|}

lib.c:dav1d_alloc_aligned_internal:
   89|  28.2k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|  28.2k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 28.2k, False: 0]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|  28.2k|    void *ptr;
   95|  28.2k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 28.2k]
  ------------------
   96|  28.2k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|  28.2k|}
lib.c:dav1d_free_aligned_internal:
  113|   367k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|   367k|}
lib.c:dav1d_freep_aligned:
  144|  9.41k|static inline void dav1d_freep_aligned(void *ptr) {
  145|  9.41k|    void **mem = (void **) ptr;
  146|  9.41k|    if (*mem) {
  ------------------
  |  Branch (146:9): [True: 9.41k, False: 0]
  ------------------
  147|  9.41k|        dav1d_free_aligned(*mem);
  ------------------
  |  |  136|  9.41k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  148|       |        *mem = NULL;
  149|  9.41k|    }
  150|  9.41k|}
mem.c:dav1d_free_aligned_internal:
  113|   118k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|   118k|}
mem.c:dav1d_alloc_aligned_internal:
   89|   118k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|   118k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 118k, False: 0]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|   118k|    void *ptr;
   95|   118k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 118k]
  ------------------
   96|   118k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|   118k|}
ref.c:dav1d_alloc_aligned_internal:
   89|   412k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|   412k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 412k, False: 0]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|   412k|    void *ptr;
   95|   412k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 412k]
  ------------------
   96|   412k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|   412k|}
ref.c:dav1d_free_aligned_internal:
  113|   412k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|   412k|}
refmvs.c:dav1d_free_aligned_internal:
  113|  30.5k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|  30.5k|}
refmvs.c:dav1d_alloc_aligned_internal:
   89|  30.5k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|  30.5k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 30.5k, False: 0]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|  30.5k|    void *ptr;
   95|  30.5k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 30.5k]
  ------------------
   96|  30.5k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|  30.5k|}
decode.c:dav1d_free_aligned_internal:
  113|   198k|static inline void dav1d_free_aligned_internal(void *ptr) {
  114|       |#ifdef _WIN32
  115|       |    _aligned_free(ptr);
  116|       |#elif HAVE_POSIX_MEMALIGN || HAVE_MEMALIGN || HAVE_ALIGNED_ALLOC
  117|       |    free(ptr);
  118|       |#else
  119|       |    if (ptr) free(((void **)ptr)[-1]);
  120|       |#endif
  121|   198k|}
decode.c:dav1d_alloc_aligned_internal:
   89|   194k|static inline void *dav1d_alloc_aligned_internal(const size_t sz, const size_t align) {
   90|   194k|    assert(!(align & (align - 1)));
  ------------------
  |  Branch (90:5): [True: 194k, False: 18.4E]
  ------------------
   91|       |#ifdef _WIN32
   92|       |    return _aligned_malloc(sz, align);
   93|       |#elif HAVE_POSIX_MEMALIGN
   94|   194k|    void *ptr;
   95|   194k|    if (posix_memalign(&ptr, align, sz)) return NULL;
  ------------------
  |  Branch (95:9): [True: 0, False: 194k]
  ------------------
   96|   194k|    return ptr;
   97|       |#elif HAVE_MEMALIGN
   98|       |    return memalign(align, sz);
   99|       |#elif HAVE_ALIGNED_ALLOC
  100|       |    // The C11 standard specifies that the size parameter
  101|       |    // must be an integral multiple of alignment.
  102|       |    return aligned_alloc(align, ROUND_UP(sz, align));
  103|       |#else
  104|       |    void *const buf = malloc(sz + align + sizeof(void *));
  105|       |    if (!buf) return NULL;
  106|       |
  107|       |    void *const ptr = (void *)(((uintptr_t)buf + sizeof(void *) + align - 1) & ~(align - 1));
  108|       |    ((void **)ptr)[-1] = buf;
  109|       |    return ptr;
  110|       |#endif
  111|   194k|}
decode.c:dav1d_freep_aligned:
  144|  3.89k|static inline void dav1d_freep_aligned(void *ptr) {
  145|  3.89k|    void **mem = (void **) ptr;
  146|  3.89k|    if (*mem) {
  ------------------
  |  Branch (146:9): [True: 3.89k, False: 0]
  ------------------
  147|  3.89k|        dav1d_free_aligned(*mem);
  ------------------
  |  |  136|  3.89k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  148|       |        *mem = NULL;
  149|  3.89k|    }
  150|  3.89k|}

dav1d_msac_decode_subexp:
   62|   231k|{
   63|   231k|    assert(n >> k == 8);
  ------------------
  |  Branch (63:5): [True: 231k, False: 18.4E]
  ------------------
   64|       |
   65|   231k|    unsigned a = 0;
   66|   231k|    if (dav1d_msac_decode_bool_equi(s)) {
  ------------------
  |  |   53|   231k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (66:9): [True: 148k, False: 83.2k]
  ------------------
   67|   148k|        if (dav1d_msac_decode_bool_equi(s))
  ------------------
  |  |   53|   148k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (67:13): [True: 104k, False: 43.4k]
  ------------------
   68|   104k|            k += dav1d_msac_decode_bool_equi(s) + 1;
  ------------------
  |  |   53|   104k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
   69|   148k|        a = 1 << k;
   70|   148k|    }
   71|   231k|    const unsigned v = dav1d_msac_decode_bools(s, k) + a;
   72|   231k|    return ref * 2 <= n ? inv_recenter(ref, v) :
  ------------------
  |  Branch (72:12): [True: 134k, False: 96.6k]
  ------------------
   73|   231k|                          n - 1 - inv_recenter(n - 1 - ref, v);
   74|   231k|}
dav1d_msac_init:
  206|   315k|{
  207|   315k|    s->buf_pos = data;
  208|   315k|    s->buf_end = data + sz;
  209|   315k|    s->dif = 0;
  210|   315k|    s->rng = 0x8000;
  211|   315k|    s->cnt = -15;
  212|   315k|    s->allow_update_cdf = !disable_cdf_update_flag;
  213|   315k|    ctx_refill(s);
  214|       |
  215|   315k|#if ARCH_X86_64 && HAVE_ASM
  216|   315k|    s->symbol_adapt16 = dav1d_msac_decode_symbol_adapt_c;
  217|       |
  218|   315k|    msac_init_x86(s);
  219|   315k|#endif
  220|   315k|}
msac.c:ctx_refill:
   41|   315k|static inline void ctx_refill(MsacContext *const s) {
   42|   315k|    const uint8_t *buf_pos = s->buf_pos;
   43|   315k|    const uint8_t *buf_end = s->buf_end;
   44|   315k|    int c = EC_WIN_SIZE - s->cnt - 24;
  ------------------
  |  |   39|   315k|#define EC_WIN_SIZE (sizeof(ec_win) << 3)
  ------------------
   45|   315k|    ec_win dif = s->dif;
   46|   973k|    do {
   47|   973k|        if (buf_pos >= buf_end) {
  ------------------
  |  Branch (47:13): [True: 287k, False: 685k]
  ------------------
   48|       |            // set remaining bits to 1;
   49|   287k|            dif |= ~(~(ec_win)0xff << c);
   50|   287k|            break;
   51|   287k|        }
   52|   685k|        dif |= (ec_win)(*buf_pos++ ^ 0xff) << c;
   53|   685k|        c -= 8;
   54|   685k|    } while (c >= 0);
  ------------------
  |  Branch (54:14): [True: 657k, False: 27.6k]
  ------------------
   55|   315k|    s->dif = dif;
   56|   315k|    s->cnt = EC_WIN_SIZE - c - 24;
  ------------------
  |  |   39|   315k|#define EC_WIN_SIZE (sizeof(ec_win) << 3)
  ------------------
   57|   315k|    s->buf_pos = buf_pos;
   58|   315k|}

decode.c:dav1d_msac_decode_bools:
   94|  1.45M|static inline unsigned dav1d_msac_decode_bools(MsacContext *const s, unsigned n) {
   95|  1.45M|    unsigned v = 0;
   96|  3.10M|    while (n--)
  ------------------
  |  Branch (96:12): [True: 1.64M, False: 1.45M]
  ------------------
   97|  1.64M|        v = (v << 1) | dav1d_msac_decode_bool_equi(s);
  ------------------
  |  |   53|  1.64M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
   98|  1.45M|    return v;
   99|  1.45M|}
decode.c:dav1d_msac_decode_uniform:
  101|   199k|static inline int dav1d_msac_decode_uniform(MsacContext *const s, const unsigned n) {
  102|   199k|    assert(n > 0);
  ------------------
  |  Branch (102:5): [True: 199k, False: 18.4E]
  ------------------
  103|   199k|    const int l = ulog2(n) + 1;
  104|   199k|    assert(l > 1);
  ------------------
  |  Branch (104:5): [True: 199k, False: 5]
  ------------------
  105|   199k|    const unsigned m = (1 << l) - n;
  106|   199k|    const unsigned v = dav1d_msac_decode_bools(s, l - 1);
  107|   199k|    return v < m ? v : (v << 1) - m + dav1d_msac_decode_bool_equi(s);
  ------------------
  |  |   53|  56.9k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (107:12): [True: 142k, False: 56.9k]
  ------------------
  108|   199k|}
msac.c:dav1d_msac_decode_bools:
   94|   231k|static inline unsigned dav1d_msac_decode_bools(MsacContext *const s, unsigned n) {
   95|   231k|    unsigned v = 0;
   96|  1.06M|    while (n--)
  ------------------
  |  Branch (96:12): [True: 828k, False: 231k]
  ------------------
   97|   828k|        v = (v << 1) | dav1d_msac_decode_bool_equi(s);
  ------------------
  |  |   53|   828k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
   98|   231k|    return v;
   99|   231k|}
recon_tmpl.c:dav1d_msac_decode_bools:
   94|  13.7M|static inline unsigned dav1d_msac_decode_bools(MsacContext *const s, unsigned n) {
   95|  13.7M|    unsigned v = 0;
   96|  44.2M|    while (n--)
  ------------------
  |  Branch (96:12): [True: 30.4M, False: 13.7M]
  ------------------
   97|  30.4M|        v = (v << 1) | dav1d_msac_decode_bool_equi(s);
  ------------------
  |  |   53|  30.4M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
   98|  13.7M|    return v;
   99|  13.7M|}

dav1d_parse_sequence_header:
  304|  15.3k|{
  305|  15.3k|    validate_input_or_ret(out != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  15.3k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 15.3k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  306|  15.3k|    validate_input_or_ret(ptr != NULL, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  15.3k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:9): [True: 0, False: 15.3k]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  307|  15.3k|    validate_input_or_ret(sz > 0 && sz <= SIZE_MAX / 2, DAV1D_ERR(EINVAL));
  ------------------
  |  |   52|  30.6k|    if (!(x)) { \
  |  |  ------------------
  |  |  |  Branch (52:11): [True: 15.3k, False: 0]
  |  |  |  Branch (52:11): [True: 15.3k, False: 0]
  |  |  ------------------
  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  ------------------
  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  ------------------
  |  |   54|      0|                    #x, __func__); \
  |  |   55|      0|        debug_abort(); \
  |  |  ------------------
  |  |  |  |   39|      0|#define debug_abort abort
  |  |  ------------------
  |  |   56|      0|        return r; \
  |  |   57|      0|    }
  ------------------
  308|       |
  309|  15.3k|    GetBits gb;
  310|  15.3k|    dav1d_init_get_bits(&gb, ptr, sz);
  311|  15.3k|    int res = DAV1D_ERR(ENOENT);
  ------------------
  |  |   58|  15.3k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  312|       |
  313|  33.4k|    do {
  314|  33.4k|        dav1d_get_bit(&gb); // obu_forbidden_bit
  315|  33.4k|        const enum Dav1dObuType type = dav1d_get_bits(&gb, 4);
  316|  33.4k|        const int has_extension = dav1d_get_bit(&gb);
  317|  33.4k|        const int has_length_field = dav1d_get_bit(&gb);
  318|  33.4k|        dav1d_get_bits(&gb, 1 + 8 * has_extension); // ignore
  319|       |
  320|  33.4k|        const uint8_t *obu_end = gb.ptr_end;
  321|  33.4k|        if (has_length_field) {
  ------------------
  |  Branch (321:13): [True: 22.0k, False: 11.3k]
  ------------------
  322|  22.0k|            const size_t len = dav1d_get_uleb128(&gb);
  323|  22.0k|            if (len > (size_t)(obu_end - gb.ptr)) return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|    537|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (323:17): [True: 537, False: 21.5k]
  ------------------
  324|  21.5k|            obu_end = gb.ptr + len;
  325|  21.5k|        }
  326|       |
  327|  32.9k|        if (type == DAV1D_OBU_SEQ_HDR) {
  ------------------
  |  Branch (327:13): [True: 12.1k, False: 20.7k]
  ------------------
  328|  12.1k|            if ((res = parse_seq_hdr(out, &gb, 0)) < 0) return res;
  ------------------
  |  Branch (328:17): [True: 2.42k, False: 9.74k]
  ------------------
  329|  9.74k|            if (gb.ptr > obu_end) return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|    223|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (329:17): [True: 223, False: 9.52k]
  ------------------
  330|  9.52k|            dav1d_bytealign_get_bits(&gb);
  331|  9.52k|        }
  332|       |
  333|  30.2k|        if (gb.error) return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|    381|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (333:13): [True: 381, False: 29.8k]
  ------------------
  334|  30.2k|        assert(gb.state == 0 && gb.bits_left == 0);
  ------------------
  |  Branch (334:9): [True: 29.8k, False: 0]
  |  Branch (334:9): [True: 29.8k, False: 0]
  ------------------
  335|  29.8k|        gb.ptr = obu_end;
  336|  29.8k|    } while (gb.ptr < gb.ptr_end);
  ------------------
  |  Branch (336:14): [True: 18.1k, False: 11.7k]
  ------------------
  337|       |
  338|  11.7k|    return res;
  339|  15.3k|}
dav1d_parse_obus:
 1169|   534k|ptrdiff_t dav1d_parse_obus(Dav1dContext *const c, Dav1dData *const in) {
 1170|   534k|    GetBits gb;
 1171|   534k|    int res;
 1172|       |
 1173|   534k|    dav1d_init_get_bits(&gb, in->data, in->sz);
 1174|       |
 1175|       |    // obu header
 1176|   534k|    const int obu_forbidden_bit = dav1d_get_bit(&gb);
 1177|   534k|    if (c->strict_std_compliance && obu_forbidden_bit) goto error;
  ------------------
  |  Branch (1177:9): [True: 0, False: 534k]
  |  Branch (1177:37): [True: 0, False: 0]
  ------------------
 1178|   534k|    const enum Dav1dObuType type = dav1d_get_bits(&gb, 4);
 1179|   534k|    const int has_extension = dav1d_get_bit(&gb);
 1180|   534k|    const int has_length_field = dav1d_get_bit(&gb);
 1181|   534k|    dav1d_get_bit(&gb); // reserved
 1182|       |
 1183|   534k|    int temporal_id = 0, spatial_id = 0;
 1184|   534k|    if (has_extension) {
  ------------------
  |  Branch (1184:9): [True: 15.1k, False: 519k]
  ------------------
 1185|  15.1k|        temporal_id = dav1d_get_bits(&gb, 3);
 1186|  15.1k|        spatial_id = dav1d_get_bits(&gb, 2);
 1187|  15.1k|        dav1d_get_bits(&gb, 3); // reserved
 1188|  15.1k|    }
 1189|       |
 1190|   534k|    if (has_length_field) {
  ------------------
  |  Branch (1190:9): [True: 159k, False: 374k]
  ------------------
 1191|   159k|        const size_t len = dav1d_get_uleb128(&gb);
 1192|   159k|        if (len > (size_t)(gb.ptr_end - gb.ptr)) goto error;
  ------------------
  |  Branch (1192:13): [True: 5.83k, False: 154k]
  ------------------
 1193|   154k|        gb.ptr_end = gb.ptr + len;
 1194|   154k|    }
 1195|   528k|    if (gb.error) goto error;
  ------------------
  |  Branch (1195:9): [True: 483, False: 528k]
  ------------------
 1196|       |
 1197|       |    // We must have read a whole number of bytes at this point (1 byte
 1198|       |    // for the header and whole bytes at a time when reading the
 1199|       |    // leb128 length field).
 1200|   528k|    assert(gb.bits_left == 0);
  ------------------
  |  Branch (1200:5): [True: 528k, False: 0]
  ------------------
 1201|       |
 1202|       |    // skip obu not belonging to the selected temporal/spatial layer
 1203|   528k|    if (type != DAV1D_OBU_SEQ_HDR && type != DAV1D_OBU_TD &&
  ------------------
  |  Branch (1203:9): [True: 496k, False: 31.8k]
  |  Branch (1203:38): [True: 434k, False: 61.4k]
  ------------------
 1204|   434k|        has_extension && c->operating_point_idc != 0)
  ------------------
  |  Branch (1204:9): [True: 12.4k, False: 422k]
  |  Branch (1204:26): [True: 5.64k, False: 6.85k]
  ------------------
 1205|  5.64k|    {
 1206|  5.64k|        const int in_temporal_layer = (c->operating_point_idc >> temporal_id) & 1;
 1207|  5.64k|        const int in_spatial_layer = (c->operating_point_idc >> (spatial_id + 8)) & 1;
 1208|  5.64k|        if (!in_temporal_layer || !in_spatial_layer)
  ------------------
  |  Branch (1208:13): [True: 514, False: 5.12k]
  |  Branch (1208:35): [True: 406, False: 4.72k]
  ------------------
 1209|    920|            return gb.ptr_end - gb.ptr_start;
 1210|  5.64k|    }
 1211|       |
 1212|   527k|    switch (type) {
 1213|  31.8k|    case DAV1D_OBU_SEQ_HDR: {
  ------------------
  |  Branch (1213:5): [True: 31.8k, False: 495k]
  ------------------
 1214|  31.8k|        Dav1dRef *ref = dav1d_ref_create_using_pool(c->seq_hdr_pool,
 1215|  31.8k|                                                    sizeof(Dav1dSequenceHeader));
 1216|  31.8k|        if (!ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (1216:13): [True: 0, False: 31.8k]
  ------------------
 1217|  31.8k|        Dav1dSequenceHeader *seq_hdr = ref->data;
 1218|  31.8k|        if ((res = parse_seq_hdr(seq_hdr, &gb, c->strict_std_compliance)) < 0) {
  ------------------
  |  Branch (1218:13): [True: 4.11k, False: 27.7k]
  ------------------
 1219|  4.11k|            dav1d_log(c, "Error parsing sequence header\n");
  ------------------
  |  |   44|  4.11k|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 4.11k]
  |  |  ------------------
  ------------------
 1220|  4.11k|            dav1d_ref_dec(&ref);
 1221|  4.11k|            goto error;
 1222|  4.11k|        }
 1223|       |
 1224|  27.7k|        const int op_idx =
 1225|  27.7k|            c->operating_point < seq_hdr->num_operating_points ? c->operating_point : 0;
  ------------------
  |  Branch (1225:13): [True: 27.7k, False: 0]
  ------------------
 1226|  27.7k|        c->operating_point_idc = seq_hdr->operating_points[op_idx].idc;
 1227|  27.7k|        const unsigned spatial_mask = c->operating_point_idc >> 8;
 1228|  27.7k|        c->max_spatial_id = spatial_mask ? ulog2(spatial_mask) : 0;
  ------------------
  |  Branch (1228:29): [True: 7.45k, False: 20.3k]
  ------------------
 1229|       |
 1230|       |        // If we have read a sequence header which is different from
 1231|       |        // the old one, this is a new video sequence and can't use any
 1232|       |        // previous state. Free that state.
 1233|       |
 1234|  27.7k|        if (!c->seq_hdr) {
  ------------------
  |  Branch (1234:13): [True: 9.12k, False: 18.6k]
  ------------------
 1235|  9.12k|            c->frame_hdr = NULL;
 1236|  9.12k|            c->frame_flags |= PICTURE_FLAG_NEW_SEQUENCE;
 1237|       |        // see 7.5, operating_parameter_info is allowed to change in
 1238|       |        // sequence headers of a single sequence
 1239|  18.6k|        } else if (memcmp(seq_hdr, c->seq_hdr, offsetof(Dav1dSequenceHeader, operating_parameter_info))) {
  ------------------
  |  Branch (1239:20): [True: 7.59k, False: 11.0k]
  ------------------
 1240|  7.59k|            c->frame_hdr = NULL;
 1241|  7.59k|            c->mastering_display = NULL;
 1242|  7.59k|            c->content_light = NULL;
 1243|  7.59k|            dav1d_ref_dec(&c->mastering_display_ref);
 1244|  7.59k|            dav1d_ref_dec(&c->content_light_ref);
 1245|  68.3k|            for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (1245:29): [True: 60.7k, False: 7.59k]
  ------------------
 1246|  60.7k|                if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (1246:21): [True: 40.7k, False: 20.0k]
  ------------------
 1247|  40.7k|                    dav1d_thread_picture_unref(&c->refs[i].p);
 1248|  60.7k|                dav1d_ref_dec(&c->refs[i].segmap);
 1249|  60.7k|                dav1d_ref_dec(&c->refs[i].refmvs);
 1250|  60.7k|                dav1d_cdf_thread_unref(&c->cdf[i]);
 1251|  60.7k|            }
 1252|  7.59k|            c->frame_flags |= PICTURE_FLAG_NEW_SEQUENCE;
 1253|       |        // If operating_parameter_info changed, signal it
 1254|  11.0k|        } else if (memcmp(seq_hdr->operating_parameter_info, c->seq_hdr->operating_parameter_info,
  ------------------
  |  Branch (1254:20): [True: 122, False: 10.9k]
  ------------------
 1255|  11.0k|                          sizeof(seq_hdr->operating_parameter_info)))
 1256|    122|        {
 1257|    122|            c->frame_flags |= PICTURE_FLAG_NEW_OP_PARAMS_INFO;
 1258|    122|        }
 1259|  27.7k|        dav1d_ref_dec(&c->seq_hdr_ref);
 1260|  27.7k|        c->seq_hdr_ref = ref;
 1261|  27.7k|        c->seq_hdr = seq_hdr;
 1262|  27.7k|        break;
 1263|  31.8k|    }
 1264|  1.40k|    case DAV1D_OBU_REDUNDANT_FRAME_HDR:
  ------------------
  |  Branch (1264:5): [True: 1.40k, False: 525k]
  ------------------
 1265|  1.40k|        if (c->frame_hdr) break;
  ------------------
  |  Branch (1265:13): [True: 509, False: 896]
  ------------------
 1266|       |        // fall-through
 1267|   381k|    case DAV1D_OBU_FRAME:
  ------------------
  |  Branch (1267:5): [True: 380k, False: 146k]
  ------------------
 1268|   417k|    case DAV1D_OBU_FRAME_HDR:
  ------------------
  |  Branch (1268:5): [True: 36.3k, False: 490k]
  ------------------
 1269|   417k|        if (!c->seq_hdr) goto error;
  ------------------
  |  Branch (1269:13): [True: 207, False: 417k]
  ------------------
 1270|   417k|        if (!c->frame_hdr_ref) {
  ------------------
  |  Branch (1270:13): [True: 360k, False: 57.1k]
  ------------------
 1271|   360k|            c->frame_hdr_ref = dav1d_ref_create_using_pool(c->frame_hdr_pool,
 1272|   360k|                                                           sizeof(Dav1dFrameHeader));
 1273|   360k|            if (!c->frame_hdr_ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (1273:17): [True: 0, False: 360k]
  ------------------
 1274|   360k|        }
 1275|   417k|#ifndef NDEBUG
 1276|       |        // ensure that the reference is writable
 1277|   417k|        assert(dav1d_ref_is_writable(c->frame_hdr_ref));
  ------------------
  |  Branch (1277:9): [True: 417k, False: 0]
  ------------------
 1278|   417k|#endif
 1279|   417k|        c->frame_hdr = c->frame_hdr_ref->data;
 1280|   417k|        memset(c->frame_hdr, 0, sizeof(*c->frame_hdr));
 1281|   417k|        c->frame_hdr->temporal_id = temporal_id;
 1282|   417k|        c->frame_hdr->spatial_id = spatial_id;
 1283|   417k|        if ((res = parse_frame_hdr(c, &gb)) < 0) {
  ------------------
  |  Branch (1283:13): [True: 7.38k, False: 410k]
  ------------------
 1284|  7.38k|            c->frame_hdr = NULL;
 1285|  7.38k|            goto error;
 1286|  7.38k|        }
 1287|   413k|        for (int n = 0; n < c->n_tile_data; n++)
  ------------------
  |  Branch (1287:25): [True: 3.03k, False: 410k]
  ------------------
 1288|  3.03k|            dav1d_data_unref_internal(&c->tile[n].data);
 1289|   410k|        c->n_tile_data = 0;
 1290|   410k|        c->n_tiles = 0;
 1291|   410k|        if (type != DAV1D_OBU_FRAME) {
  ------------------
  |  Branch (1291:13): [True: 35.9k, False: 374k]
  ------------------
 1292|       |            // This is actually a frame header OBU so read the
 1293|       |            // trailing bit and check for overrun.
 1294|  35.9k|            if (check_trailing_bits(&gb, c->strict_std_compliance) < 0) {
  ------------------
  |  Branch (1294:17): [True: 2.34k, False: 33.6k]
  ------------------
 1295|  2.34k|                c->frame_hdr = NULL;
 1296|  2.34k|                goto error;
 1297|  2.34k|            }
 1298|  35.9k|        }
 1299|       |
 1300|   407k|        if (c->frame_size_limit && (int64_t)c->frame_hdr->width[1] *
  ------------------
  |  Branch (1300:13): [True: 407k, False: 0]
  |  Branch (1300:36): [True: 420, False: 407k]
  ------------------
 1301|   407k|            c->frame_hdr->height > c->frame_size_limit)
 1302|    420|        {
 1303|    420|            dav1d_log(c, "Frame size %dx%d exceeds limit %u\n", c->frame_hdr->width[1],
  ------------------
  |  |   44|    420|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 420]
  |  |  ------------------
  ------------------
 1304|    420|                      c->frame_hdr->height, c->frame_size_limit);
 1305|    420|            c->frame_hdr = NULL;
 1306|    420|            return DAV1D_ERR(ERANGE);
  ------------------
  |  |   58|    420|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 1307|    420|        }
 1308|       |
 1309|   407k|        if (type != DAV1D_OBU_FRAME)
  ------------------
  |  Branch (1309:13): [True: 33.6k, False: 373k]
  ------------------
 1310|  33.6k|            break;
 1311|       |        // OBU_FRAMEs shouldn't be signaled with show_existing_frame
 1312|   373k|        if (c->frame_hdr->show_existing_frame) {
  ------------------
  |  Branch (1312:13): [True: 548, False: 373k]
  ------------------
 1313|    548|            c->frame_hdr = NULL;
 1314|    548|            goto error;
 1315|    548|        }
 1316|       |
 1317|       |        // This is the frame header at the start of a frame OBU.
 1318|       |        // There's no trailing bit at the end to skip, but we do need
 1319|       |        // to align to the next byte.
 1320|   373k|        dav1d_bytealign_get_bits(&gb);
 1321|       |        // fall-through
 1322|   375k|    case DAV1D_OBU_TILE_GRP: {
  ------------------
  |  Branch (1322:5): [True: 2.35k, False: 524k]
  ------------------
 1323|   375k|        if (!c->frame_hdr) goto error;
  ------------------
  |  Branch (1323:13): [True: 1.48k, False: 374k]
  ------------------
 1324|   374k|        if (c->n_tile_data_alloc < c->n_tile_data + 1) {
  ------------------
  |  Branch (1324:13): [True: 8.74k, False: 365k]
  ------------------
 1325|  8.74k|            if ((c->n_tile_data + 1) > INT_MAX / (int)sizeof(*c->tile)) goto error;
  ------------------
  |  Branch (1325:17): [True: 0, False: 8.74k]
  ------------------
 1326|  8.74k|            struct Dav1dTileGroup *tile = dav1d_realloc(ALLOC_TILE, c->tile,
  ------------------
  |  |  133|  8.74k|#define dav1d_realloc(type, ptr, sz) realloc(ptr, sz)
  ------------------
 1327|  8.74k|                                                        (c->n_tile_data + 1) * sizeof(*c->tile));
 1328|  8.74k|            if (!tile) goto error;
  ------------------
  |  Branch (1328:17): [True: 0, False: 8.74k]
  ------------------
 1329|  8.74k|            c->tile = tile;
 1330|  8.74k|            memset(c->tile + c->n_tile_data, 0, sizeof(*c->tile));
 1331|  8.74k|            c->n_tile_data_alloc = c->n_tile_data + 1;
 1332|  8.74k|        }
 1333|   374k|        parse_tile_hdr(c, &gb);
 1334|       |        // Align to the next byte boundary and check for overrun.
 1335|   374k|        dav1d_bytealign_get_bits(&gb);
 1336|   374k|        if (gb.error) goto error;
  ------------------
  |  Branch (1336:13): [True: 12.7k, False: 361k]
  ------------------
 1337|       |
 1338|   361k|        dav1d_data_ref(&c->tile[c->n_tile_data].data, in);
 1339|   361k|        c->tile[c->n_tile_data].data.data = gb.ptr;
 1340|   361k|        c->tile[c->n_tile_data].data.sz = (size_t)(gb.ptr_end - gb.ptr);
 1341|       |        // ensure tile groups are in order and sane, see 6.10.1
 1342|   361k|        if (c->tile[c->n_tile_data].start > c->tile[c->n_tile_data].end ||
  ------------------
  |  Branch (1342:13): [True: 946, False: 360k]
  ------------------
 1343|   360k|            c->tile[c->n_tile_data].start != c->n_tiles)
  ------------------
  |  Branch (1343:13): [True: 625, False: 359k]
  ------------------
 1344|  1.57k|        {
 1345|  3.30k|            for (int i = 0; i <= c->n_tile_data; i++)
  ------------------
  |  Branch (1345:29): [True: 1.73k, False: 1.57k]
  ------------------
 1346|  1.73k|                dav1d_data_unref_internal(&c->tile[i].data);
 1347|  1.57k|            c->n_tile_data = 0;
 1348|  1.57k|            c->n_tiles = 0;
 1349|  1.57k|            goto error;
 1350|  1.57k|        }
 1351|   359k|        c->n_tiles += 1 + c->tile[c->n_tile_data].end -
 1352|   359k|                          c->tile[c->n_tile_data].start;
 1353|   359k|        c->n_tile_data++;
 1354|   359k|        break;
 1355|   361k|    }
 1356|  5.84k|    case DAV1D_OBU_METADATA: {
  ------------------
  |  Branch (1356:5): [True: 5.84k, False: 521k]
  ------------------
 1357|  5.84k|#define DEBUG_OBU_METADATA 0
 1358|       |#if DEBUG_OBU_METADATA
 1359|       |        const uint8_t *const init_ptr = gb.ptr;
 1360|       |#endif
 1361|       |        // obu metadta type field
 1362|  5.84k|        const enum ObuMetaType meta_type = dav1d_get_uleb128(&gb);
 1363|  5.84k|        if (gb.error) goto error;
  ------------------
  |  Branch (1363:13): [True: 284, False: 5.55k]
  ------------------
 1364|       |
 1365|  5.55k|        switch (meta_type) {
 1366|    653|        case OBU_META_HDR_CLL: {
  ------------------
  |  Branch (1366:9): [True: 653, False: 4.90k]
  ------------------
 1367|    653|            Dav1dRef *ref = dav1d_ref_create(ALLOC_OBU_META,
  ------------------
  |  |   49|    653|#define dav1d_ref_create(type, size) dav1d_ref_create(size)
  ------------------
 1368|    653|                                             sizeof(Dav1dContentLightLevel));
 1369|    653|            if (!ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (1369:17): [True: 0, False: 653]
  ------------------
 1370|    653|            Dav1dContentLightLevel *const content_light = ref->data;
 1371|       |
 1372|    653|            content_light->max_content_light_level = dav1d_get_bits(&gb, 16);
 1373|       |#if DEBUG_OBU_METADATA
 1374|       |            printf("CLLOBU: max-content-light-level: %d [off=%td]\n",
 1375|       |                   content_light->max_content_light_level,
 1376|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1377|       |#endif
 1378|    653|            content_light->max_frame_average_light_level = dav1d_get_bits(&gb, 16);
 1379|       |#if DEBUG_OBU_METADATA
 1380|       |            printf("CLLOBU: max-frame-average-light-level: %d [off=%td]\n",
 1381|       |                   content_light->max_frame_average_light_level,
 1382|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1383|       |#endif
 1384|       |
 1385|    653|            if (check_trailing_bits(&gb, c->strict_std_compliance) < 0) {
  ------------------
  |  Branch (1385:17): [True: 207, False: 446]
  ------------------
 1386|    207|                dav1d_ref_dec(&ref);
 1387|    207|                goto error;
 1388|    207|            }
 1389|       |
 1390|    446|            dav1d_ref_dec(&c->content_light_ref);
 1391|    446|            c->content_light = content_light;
 1392|    446|            c->content_light_ref = ref;
 1393|    446|            break;
 1394|    653|        }
 1395|    237|        case OBU_META_HDR_MDCV: {
  ------------------
  |  Branch (1395:9): [True: 237, False: 5.31k]
  ------------------
 1396|    237|            Dav1dRef *ref = dav1d_ref_create(ALLOC_OBU_META,
  ------------------
  |  |   49|    237|#define dav1d_ref_create(type, size) dav1d_ref_create(size)
  ------------------
 1397|    237|                                             sizeof(Dav1dMasteringDisplay));
 1398|    237|            if (!ref) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (1398:17): [True: 0, False: 237]
  ------------------
 1399|    237|            Dav1dMasteringDisplay *const mastering_display = ref->data;
 1400|       |
 1401|    948|            for (int i = 0; i < 3; i++) {
  ------------------
  |  Branch (1401:29): [True: 711, False: 237]
  ------------------
 1402|    711|                mastering_display->primaries[i][0] = dav1d_get_bits(&gb, 16);
 1403|    711|                mastering_display->primaries[i][1] = dav1d_get_bits(&gb, 16);
 1404|       |#if DEBUG_OBU_METADATA
 1405|       |                printf("MDCVOBU: primaries[%d]: (%d, %d) [off=%td]\n", i,
 1406|       |                       mastering_display->primaries[i][0],
 1407|       |                       mastering_display->primaries[i][1],
 1408|       |                       (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1409|       |#endif
 1410|    711|            }
 1411|    237|            mastering_display->white_point[0] = dav1d_get_bits(&gb, 16);
 1412|       |#if DEBUG_OBU_METADATA
 1413|       |            printf("MDCVOBU: white-point-x: %d [off=%td]\n",
 1414|       |                   mastering_display->white_point[0],
 1415|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1416|       |#endif
 1417|    237|            mastering_display->white_point[1] = dav1d_get_bits(&gb, 16);
 1418|       |#if DEBUG_OBU_METADATA
 1419|       |            printf("MDCVOBU: white-point-y: %d [off=%td]\n",
 1420|       |                   mastering_display->white_point[1],
 1421|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1422|       |#endif
 1423|    237|            mastering_display->max_luminance = dav1d_get_bits(&gb, 32);
 1424|       |#if DEBUG_OBU_METADATA
 1425|       |            printf("MDCVOBU: max-luminance: %d [off=%td]\n",
 1426|       |                   mastering_display->max_luminance,
 1427|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1428|       |#endif
 1429|    237|            mastering_display->min_luminance = dav1d_get_bits(&gb, 32);
 1430|       |#if DEBUG_OBU_METADATA
 1431|       |            printf("MDCVOBU: min-luminance: %d [off=%td]\n",
 1432|       |                   mastering_display->min_luminance,
 1433|       |                   (gb.ptr - init_ptr) * 8 - gb.bits_left);
 1434|       |#endif
 1435|    237|            if (check_trailing_bits(&gb, c->strict_std_compliance) < 0) {
  ------------------
  |  Branch (1435:17): [True: 153, False: 84]
  ------------------
 1436|    153|                dav1d_ref_dec(&ref);
 1437|    153|                goto error;
 1438|    153|            }
 1439|       |
 1440|     84|            dav1d_ref_dec(&c->mastering_display_ref);
 1441|     84|            c->mastering_display = mastering_display;
 1442|     84|            c->mastering_display_ref = ref;
 1443|     84|            break;
 1444|    237|        }
 1445|  4.24k|        case OBU_META_ITUT_T35: {
  ------------------
  |  Branch (1445:9): [True: 4.24k, False: 1.30k]
  ------------------
 1446|  4.24k|            ptrdiff_t payload_size = gb.ptr_end - gb.ptr;
 1447|       |            // Don't take into account all the trailing bits for payload_size
 1448|  4.53k|            while (payload_size > 0 && !gb.ptr[payload_size - 1])
  ------------------
  |  Branch (1448:20): [True: 3.87k, False: 662]
  |  Branch (1448:40): [True: 286, False: 3.58k]
  ------------------
 1449|    286|                payload_size--; // trailing_zero_bit x 8
 1450|  4.24k|            payload_size--; // trailing_one_bit + trailing_zero_bit x 7
 1451|       |
 1452|  4.24k|            int country_code_extension_byte = 0;
 1453|  4.24k|            const int country_code = dav1d_get_bits(&gb, 8);
 1454|  4.24k|            payload_size--;
 1455|  4.24k|            if (country_code == 0xFF) {
  ------------------
  |  Branch (1455:17): [True: 1.86k, False: 2.38k]
  ------------------
 1456|  1.86k|                country_code_extension_byte = dav1d_get_bits(&gb, 8);
 1457|  1.86k|                payload_size--;
 1458|  1.86k|            }
 1459|       |
 1460|  4.24k|            if (payload_size <= 0 || gb.ptr[payload_size] != 0x80) {
  ------------------
  |  Branch (1460:17): [True: 726, False: 3.52k]
  |  Branch (1460:38): [True: 1.44k, False: 2.08k]
  ------------------
 1461|  2.16k|                dav1d_log(c, "Malformed ITU-T T.35 metadata message format\n");
  ------------------
  |  |   44|  2.16k|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 2.16k]
  |  |  ------------------
  ------------------
 1462|  2.16k|                break;
 1463|  2.16k|            }
 1464|       |
 1465|  2.08k|            if ((c->n_itut_t35 + 1) > INT_MAX / (int)sizeof(*c->itut_t35)) goto error;
  ------------------
  |  Branch (1465:17): [True: 0, False: 2.08k]
  ------------------
 1466|  2.08k|            struct Dav1dITUTT35 *itut_t35 = dav1d_realloc(ALLOC_OBU_META, c->itut_t35,
  ------------------
  |  |  133|  2.08k|#define dav1d_realloc(type, ptr, sz) realloc(ptr, sz)
  ------------------
 1467|  2.08k|                                                          (c->n_itut_t35 + 1) * sizeof(*c->itut_t35));
 1468|  2.08k|            if (!itut_t35) goto error;
  ------------------
  |  Branch (1468:17): [True: 0, False: 2.08k]
  ------------------
 1469|  2.08k|            c->itut_t35 = itut_t35;
 1470|  2.08k|            memset(c->itut_t35 + c->n_itut_t35, 0, sizeof(*c->itut_t35));
 1471|       |
 1472|  2.08k|            struct itut_t35_ctx_context *itut_t35_ctx;
 1473|  2.08k|            if (!c->n_itut_t35) {
  ------------------
  |  Branch (1473:17): [True: 972, False: 1.10k]
  ------------------
 1474|    972|                assert(!c->itut_t35_ref);
  ------------------
  |  Branch (1474:17): [True: 972, False: 0]
  ------------------
 1475|    972|                itut_t35_ctx = dav1d_malloc(ALLOC_OBU_META, sizeof(struct itut_t35_ctx_context));
  ------------------
  |  |  132|    972|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 1476|    972|                if (!itut_t35_ctx) goto error;
  ------------------
  |  Branch (1476:21): [True: 0, False: 972]
  ------------------
 1477|    972|                c->itut_t35_ref = dav1d_ref_init(&itut_t35_ctx->ref, c->itut_t35,
 1478|    972|                                                 dav1d_picture_free_itut_t35, itut_t35_ctx, 0);
 1479|  1.10k|            } else {
 1480|  1.10k|                assert(c->itut_t35_ref && atomic_load(&c->itut_t35_ref->ref_cnt) == 1);
  ------------------
  |  Branch (1480:17): [True: 1.10k, False: 0]
  |  Branch (1480:17): [True: 1.10k, False: 0]
  ------------------
 1481|  1.10k|                itut_t35_ctx = c->itut_t35_ref->user_data;
 1482|  1.10k|                c->itut_t35_ref->const_data = (uint8_t *)c->itut_t35;
 1483|  1.10k|            }
 1484|  2.08k|            itut_t35_ctx->itut_t35 = c->itut_t35;
 1485|  2.08k|            itut_t35_ctx->n_itut_t35 = c->n_itut_t35 + 1;
 1486|       |
 1487|  2.08k|            Dav1dITUTT35 *const itut_t35_metadata = &c->itut_t35[c->n_itut_t35];
 1488|  2.08k|            itut_t35_metadata->payload = dav1d_malloc(ALLOC_OBU_META, payload_size);
  ------------------
  |  |  132|  2.08k|#define dav1d_malloc(type, sz) malloc(sz)
  ------------------
 1489|  2.08k|            if (!itut_t35_metadata->payload) goto error;
  ------------------
  |  Branch (1489:17): [True: 0, False: 2.08k]
  ------------------
 1490|       |
 1491|  2.08k|            itut_t35_metadata->country_code = country_code;
 1492|  2.08k|            itut_t35_metadata->country_code_extension_byte = country_code_extension_byte;
 1493|  2.08k|            itut_t35_metadata->payload_size = payload_size;
 1494|       |
 1495|       |            // We know that we've read a whole number of bytes and that the
 1496|       |            // payload is within the OBU boundaries, so just use memcpy()
 1497|  2.08k|            assert(gb.bits_left == 0);
  ------------------
  |  Branch (1497:13): [True: 2.08k, False: 0]
  ------------------
 1498|  2.08k|            memcpy(itut_t35_metadata->payload, gb.ptr, payload_size);
 1499|       |
 1500|  2.08k|            c->n_itut_t35++;
 1501|  2.08k|            break;
 1502|  2.08k|        }
 1503|      2|        case OBU_META_SCALABILITY:
  ------------------
  |  Branch (1503:9): [True: 2, False: 5.55k]
  ------------------
 1504|     13|        case OBU_META_TIMECODE:
  ------------------
  |  Branch (1504:9): [True: 11, False: 5.54k]
  ------------------
 1505|       |            // ignore metadata OBUs we don't care about
 1506|     13|            break;
 1507|    405|        default:
  ------------------
  |  Branch (1507:9): [True: 405, False: 5.15k]
  ------------------
 1508|       |            // print a warning but don't fail for unknown types
 1509|    405|            if (meta_type > 31) // Types 6 to 31 are "Unregistered user private", so ignore them.
  ------------------
  |  Branch (1509:17): [True: 314, False: 91]
  ------------------
 1510|    314|                dav1d_log(c, "Unknown Metadata OBU type %d\n", meta_type);
  ------------------
  |  |   44|    314|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 314]
  |  |  ------------------
  ------------------
 1511|    405|            break;
 1512|  5.55k|        }
 1513|       |
 1514|  5.19k|        break;
 1515|  5.55k|    }
 1516|  61.4k|    case DAV1D_OBU_TD:
  ------------------
  |  Branch (1516:5): [True: 61.4k, False: 465k]
  ------------------
 1517|  61.4k|        c->frame_flags |= PICTURE_FLAG_NEW_TEMPORAL_UNIT;
 1518|  61.4k|        break;
 1519|    549|    case DAV1D_OBU_PADDING:
  ------------------
  |  Branch (1519:5): [True: 549, False: 526k]
  ------------------
 1520|       |        // ignore OBUs we don't care about
 1521|    549|        break;
 1522|  6.67k|    default:
  ------------------
  |  Branch (1522:5): [True: 6.67k, False: 520k]
  ------------------
 1523|       |        // print a warning but don't fail for unknown types
 1524|  6.67k|        dav1d_log(c, "Unknown OBU type %d of size %td\n", type, gb.ptr_end - gb.ptr);
  ------------------
  |  |   44|  6.67k|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 6.67k]
  |  |  ------------------
  ------------------
 1525|  6.67k|        break;
 1526|   527k|    }
 1527|       |
 1528|   495k|    if (c->seq_hdr && c->frame_hdr) {
  ------------------
  |  Branch (1528:9): [True: 492k, False: 3.54k]
  |  Branch (1528:23): [True: 402k, False: 89.7k]
  ------------------
 1529|   402k|        if (c->frame_hdr->show_existing_frame) {
  ------------------
  |  Branch (1529:13): [True: 31.7k, False: 370k]
  ------------------
 1530|  31.7k|            if (!c->refs[c->frame_hdr->existing_frame_idx].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (1530:17): [True: 283, False: 31.4k]
  ------------------
 1531|  31.4k|            switch (c->refs[c->frame_hdr->existing_frame_idx].p.p.frame_hdr->frame_type) {
 1532|  17.3k|            case DAV1D_FRAME_TYPE_INTER:
  ------------------
  |  Branch (1532:13): [True: 17.3k, False: 14.0k]
  ------------------
 1533|  17.7k|            case DAV1D_FRAME_TYPE_SWITCH:
  ------------------
  |  Branch (1533:13): [True: 391, False: 31.0k]
  ------------------
 1534|  17.7k|                if (c->decode_frame_type > DAV1D_DECODEFRAMETYPE_REFERENCE)
  ------------------
  |  Branch (1534:21): [True: 0, False: 17.7k]
  ------------------
 1535|      0|                    goto skip;
 1536|  17.7k|                break;
 1537|  17.7k|            case DAV1D_FRAME_TYPE_INTRA:
  ------------------
  |  Branch (1537:13): [True: 359, False: 31.0k]
  ------------------
 1538|    359|                if (c->decode_frame_type > DAV1D_DECODEFRAMETYPE_INTRA)
  ------------------
  |  Branch (1538:21): [True: 0, False: 359]
  ------------------
 1539|      0|                    goto skip;
 1540|       |                // fall-through
 1541|  13.6k|            default:
  ------------------
  |  Branch (1541:13): [True: 13.3k, False: 18.1k]
  ------------------
 1542|  13.6k|                break;
 1543|  31.4k|            }
 1544|  31.4k|            if (!c->refs[c->frame_hdr->existing_frame_idx].p.p.data[0]) goto error;
  ------------------
  |  Branch (1544:17): [True: 0, False: 31.4k]
  ------------------
 1545|  31.4k|            if (c->strict_std_compliance &&
  ------------------
  |  Branch (1545:17): [True: 0, False: 31.4k]
  ------------------
 1546|      0|                !c->refs[c->frame_hdr->existing_frame_idx].p.showable)
  ------------------
  |  Branch (1546:17): [True: 0, False: 0]
  ------------------
 1547|      0|            {
 1548|      0|                goto error;
 1549|      0|            }
 1550|  31.4k|            if (c->n_fc == 1) {
  ------------------
  |  Branch (1550:17): [True: 0, False: 31.4k]
  ------------------
 1551|      0|                dav1d_thread_picture_ref(&c->out,
 1552|      0|                                         &c->refs[c->frame_hdr->existing_frame_idx].p);
 1553|      0|                dav1d_picture_copy_props(&c->out.p,
 1554|      0|                                         c->content_light, c->content_light_ref,
 1555|      0|                                         c->mastering_display, c->mastering_display_ref,
 1556|      0|                                         c->itut_t35, c->itut_t35_ref, c->n_itut_t35,
 1557|      0|                                         &in->m);
 1558|       |                // Must be removed from the context after being attached to the frame
 1559|      0|                dav1d_ref_dec(&c->itut_t35_ref);
 1560|      0|                c->itut_t35 = NULL;
 1561|      0|                c->n_itut_t35 = 0;
 1562|      0|                c->event_flags |= dav1d_picture_get_event_flags(&c->refs[c->frame_hdr->existing_frame_idx].p);
 1563|  31.4k|            } else {
 1564|  31.4k|                pthread_mutex_lock(&c->task_thread.lock);
 1565|       |                // need to append this to the frame output queue
 1566|  31.4k|                const unsigned next = c->frame_thread.next++;
 1567|  31.4k|                if (c->frame_thread.next == c->n_fc)
  ------------------
  |  Branch (1567:21): [True: 7.73k, False: 23.6k]
  ------------------
 1568|  7.73k|                    c->frame_thread.next = 0;
 1569|       |
 1570|  31.4k|                Dav1dFrameContext *const f = &c->fc[next];
 1571|  34.7k|                while (f->n_tile_data > 0)
  ------------------
  |  Branch (1571:24): [True: 3.32k, False: 31.4k]
  ------------------
 1572|  3.32k|                    pthread_cond_wait(&f->task_thread.cond,
 1573|  3.32k|                                      &f->task_thread.ttd->lock);
 1574|  31.4k|                Dav1dThreadPicture *const out_delayed =
 1575|  31.4k|                    &c->frame_thread.out_delayed[next];
 1576|  31.4k|                if (out_delayed->p.data[0] || atomic_load(&f->task_thread.error)) {
  ------------------
  |  Branch (1576:21): [True: 30.6k, False: 742]
  |  Branch (1576:47): [True: 120, False: 622]
  ------------------
 1577|  30.8k|                    unsigned first = atomic_load(&c->task_thread.first);
 1578|  30.8k|                    if (first + 1U < c->n_fc)
  ------------------
  |  Branch (1578:25): [True: 23.2k, False: 7.56k]
  ------------------
 1579|  30.8k|                        atomic_fetch_add(&c->task_thread.first, 1U);
 1580|  7.56k|                    else
 1581|  30.8k|                        atomic_store(&c->task_thread.first, 0);
 1582|  30.8k|                    atomic_compare_exchange_strong(&c->task_thread.reset_task_cur,
 1583|  30.8k|                                                   &first, UINT_MAX);
 1584|  30.8k|                    if (c->task_thread.cur && c->task_thread.cur < c->n_fc)
  ------------------
  |  Branch (1584:25): [True: 30.0k, False: 753]
  |  Branch (1584:47): [True: 18.0k, False: 12.0k]
  ------------------
 1585|  18.0k|                        c->task_thread.cur--;
 1586|  30.8k|                }
 1587|  31.4k|                const int error = f->task_thread.retval;
 1588|  31.4k|                if (error) {
  ------------------
  |  Branch (1588:21): [True: 2.44k, False: 28.9k]
  ------------------
 1589|  2.44k|                    c->cached_error = error;
 1590|  2.44k|                    f->task_thread.retval = 0;
 1591|  2.44k|                    dav1d_data_props_copy(&c->cached_error_props, &out_delayed->p.m);
 1592|  2.44k|                    dav1d_thread_picture_unref(out_delayed);
 1593|  28.9k|                } else if (out_delayed->p.data[0]) {
  ------------------
  |  Branch (1593:28): [True: 28.2k, False: 742]
  ------------------
 1594|  28.2k|                    const unsigned progress = atomic_load_explicit(&out_delayed->progress[1],
 1595|  28.2k|                                                                   memory_order_relaxed);
 1596|  28.2k|                    if ((out_delayed->visible || c->output_invisible_frames) &&
  ------------------
  |  Branch (1596:26): [True: 27.9k, False: 242]
  |  Branch (1596:50): [True: 0, False: 242]
  ------------------
 1597|  27.9k|                        progress != FRAME_ERROR)
  ------------------
  |  |   35|  27.9k|#define FRAME_ERROR (UINT_MAX - 1)
  ------------------
  |  Branch (1597:25): [True: 13.7k, False: 14.2k]
  ------------------
 1598|  13.7k|                    {
 1599|  13.7k|                        dav1d_thread_picture_ref(&c->out, out_delayed);
 1600|  13.7k|                        c->event_flags |= dav1d_picture_get_event_flags(out_delayed);
 1601|  13.7k|                    }
 1602|  28.2k|                    dav1d_thread_picture_unref(out_delayed);
 1603|  28.2k|                }
 1604|  31.4k|                dav1d_thread_picture_ref(out_delayed,
 1605|  31.4k|                                         &c->refs[c->frame_hdr->existing_frame_idx].p);
 1606|  31.4k|                out_delayed->visible = 1;
 1607|  31.4k|                dav1d_picture_copy_props(&out_delayed->p,
 1608|  31.4k|                                         c->content_light, c->content_light_ref,
 1609|  31.4k|                                         c->mastering_display, c->mastering_display_ref,
 1610|  31.4k|                                         c->itut_t35, c->itut_t35_ref, c->n_itut_t35,
 1611|  31.4k|                                         &in->m);
 1612|       |                // Must be removed from the context after being attached to the frame
 1613|  31.4k|                dav1d_ref_dec(&c->itut_t35_ref);
 1614|  31.4k|                c->itut_t35 = NULL;
 1615|  31.4k|                c->n_itut_t35 = 0;
 1616|       |
 1617|  31.4k|                pthread_mutex_unlock(&c->task_thread.lock);
 1618|  31.4k|            }
 1619|  31.4k|            if (c->refs[c->frame_hdr->existing_frame_idx].p.p.frame_hdr->frame_type == DAV1D_FRAME_TYPE_KEY) {
  ------------------
  |  Branch (1619:17): [True: 13.3k, False: 18.1k]
  ------------------
 1620|  13.3k|                const int r = c->frame_hdr->existing_frame_idx;
 1621|  13.3k|                c->refs[r].p.showable = 0;
 1622|   119k|                for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (1622:33): [True: 106k, False: 13.3k]
  ------------------
 1623|   106k|                    if (i == r) continue;
  ------------------
  |  Branch (1623:25): [True: 13.3k, False: 93.2k]
  ------------------
 1624|       |
 1625|  93.2k|                    if (c->refs[i].p.p.frame_hdr)
  ------------------
  |  Branch (1625:25): [True: 93.1k, False: 114]
  ------------------
 1626|  93.1k|                        dav1d_thread_picture_unref(&c->refs[i].p);
 1627|  93.2k|                    dav1d_thread_picture_ref(&c->refs[i].p, &c->refs[r].p);
 1628|       |
 1629|  93.2k|                    dav1d_cdf_thread_unref(&c->cdf[i]);
 1630|  93.2k|                    dav1d_cdf_thread_ref(&c->cdf[i], &c->cdf[r]);
 1631|       |
 1632|  93.2k|                    dav1d_ref_dec(&c->refs[i].segmap);
 1633|  93.2k|                    c->refs[i].segmap = c->refs[r].segmap;
 1634|  93.2k|                    if (c->refs[r].segmap)
  ------------------
  |  Branch (1634:25): [True: 9.17k, False: 84.1k]
  ------------------
 1635|  9.17k|                        dav1d_ref_inc(c->refs[r].segmap);
 1636|  93.2k|                    dav1d_ref_dec(&c->refs[i].refmvs);
 1637|  93.2k|                }
 1638|  13.3k|            }
 1639|  31.4k|            c->frame_hdr = NULL;
 1640|   370k|        } else if (c->n_tiles == c->frame_hdr->tiling.cols * c->frame_hdr->tiling.rows) {
  ------------------
  |  Branch (1640:20): [True: 359k, False: 11.2k]
  ------------------
 1641|   359k|            switch (c->frame_hdr->frame_type) {
 1642|   100k|            case DAV1D_FRAME_TYPE_INTER:
  ------------------
  |  Branch (1642:13): [True: 100k, False: 258k]
  ------------------
 1643|   101k|            case DAV1D_FRAME_TYPE_SWITCH:
  ------------------
  |  Branch (1643:13): [True: 676, False: 358k]
  ------------------
 1644|   101k|                if (c->decode_frame_type > DAV1D_DECODEFRAMETYPE_REFERENCE ||
  ------------------
  |  Branch (1644:21): [True: 0, False: 101k]
  ------------------
 1645|   101k|                    (c->decode_frame_type == DAV1D_DECODEFRAMETYPE_REFERENCE &&
  ------------------
  |  Branch (1645:22): [True: 0, False: 101k]
  ------------------
 1646|      0|                     !c->frame_hdr->refresh_frame_flags))
  ------------------
  |  Branch (1646:22): [True: 0, False: 0]
  ------------------
 1647|      0|                    goto skip;
 1648|   101k|                break;
 1649|   101k|            case DAV1D_FRAME_TYPE_INTRA:
  ------------------
  |  Branch (1649:13): [True: 903, False: 358k]
  ------------------
 1650|    903|                if (c->decode_frame_type > DAV1D_DECODEFRAMETYPE_INTRA ||
  ------------------
  |  Branch (1650:21): [True: 0, False: 903]
  ------------------
 1651|    903|                    (c->decode_frame_type == DAV1D_DECODEFRAMETYPE_REFERENCE &&
  ------------------
  |  Branch (1651:22): [True: 0, False: 903]
  ------------------
 1652|      0|                     !c->frame_hdr->refresh_frame_flags))
  ------------------
  |  Branch (1652:22): [True: 0, False: 0]
  ------------------
 1653|      0|                    goto skip;
 1654|       |                // fall-through
 1655|   258k|            default:
  ------------------
  |  Branch (1655:13): [True: 257k, False: 102k]
  ------------------
 1656|   258k|                break;
 1657|   359k|            }
 1658|   359k|            if (!c->n_tile_data)
  ------------------
  |  Branch (1658:17): [True: 0, False: 359k]
  ------------------
 1659|      0|                goto error;
 1660|   359k|            if ((res = dav1d_submit_frame(c)) < 0)
  ------------------
  |  Branch (1660:17): [True: 3.01k, False: 356k]
  ------------------
 1661|  3.01k|                return res;
 1662|   359k|            assert(!c->n_tile_data);
  ------------------
  |  Branch (1662:13): [True: 356k, False: 0]
  ------------------
 1663|   356k|            c->frame_hdr = NULL;
 1664|   356k|            c->n_tiles = 0;
 1665|   356k|        }
 1666|   402k|    }
 1667|       |
 1668|   492k|    return gb.ptr_end - gb.ptr_start;
 1669|       |
 1670|      0|skip:
 1671|       |    // update refs with only the headers in case we skip the frame
 1672|      0|    for (int i = 0; i < 8; i++) {
  ------------------
  |  Branch (1672:21): [True: 0, False: 0]
  ------------------
 1673|      0|        if (c->frame_hdr->refresh_frame_flags & (1 << i)) {
  ------------------
  |  Branch (1673:13): [True: 0, False: 0]
  ------------------
 1674|      0|            dav1d_thread_picture_unref(&c->refs[i].p);
 1675|      0|            c->refs[i].p.p.frame_hdr = c->frame_hdr;
 1676|      0|            c->refs[i].p.p.seq_hdr = c->seq_hdr;
 1677|      0|            c->refs[i].p.p.frame_hdr_ref = c->frame_hdr_ref;
 1678|      0|            c->refs[i].p.p.seq_hdr_ref = c->seq_hdr_ref;
 1679|      0|            dav1d_ref_inc(c->frame_hdr_ref);
 1680|      0|            dav1d_ref_inc(c->seq_hdr_ref);
 1681|      0|        }
 1682|      0|    }
 1683|       |
 1684|      0|    dav1d_ref_dec(&c->frame_hdr_ref);
 1685|      0|    c->frame_hdr = NULL;
 1686|      0|    c->n_tiles = 0;
 1687|       |
 1688|      0|    return gb.ptr_end - gb.ptr_start;
 1689|       |
 1690|  37.6k|error:
 1691|  37.6k|    dav1d_data_props_copy(&c->cached_error_props, &in->m);
 1692|  37.6k|    dav1d_log(c, gb.error ? "Overrun in OBU bit buffer\n" :
  ------------------
  |  |   44|  37.6k|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 37.6k]
  |  |  ------------------
  ------------------
 1693|  37.6k|                            "Error parsing OBU data\n");
 1694|  37.6k|    return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  37.6k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 1695|   495k|}
obu.c:parse_seq_hdr:
   75|  44.0k|{
   76|  44.0k|#define DEBUG_SEQ_HDR 0
   77|       |
   78|       |#if DEBUG_SEQ_HDR
   79|       |    const unsigned init_bit_pos = dav1d_get_bits_pos(gb);
   80|       |#endif
   81|       |
   82|  44.0k|    memset(hdr, 0, sizeof(*hdr));
   83|  44.0k|    hdr->profile = dav1d_get_bits(gb, 3);
   84|  44.0k|    if (hdr->profile > 2) goto error;
  ------------------
  |  Branch (84:9): [True: 642, False: 43.4k]
  ------------------
   85|       |#if DEBUG_SEQ_HDR
   86|       |    printf("SEQHDR: post-profile: off=%u\n",
   87|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
   88|       |#endif
   89|       |
   90|  43.4k|    hdr->still_picture = dav1d_get_bit(gb);
   91|  43.4k|    hdr->reduced_still_picture_header = dav1d_get_bit(gb);
   92|  43.4k|    if (hdr->reduced_still_picture_header && !hdr->still_picture) goto error;
  ------------------
  |  Branch (92:9): [True: 23.4k, False: 19.9k]
  |  Branch (92:46): [True: 1.17k, False: 22.2k]
  ------------------
   93|       |#if DEBUG_SEQ_HDR
   94|       |    printf("SEQHDR: post-stillpicture_flags: off=%u\n",
   95|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
   96|       |#endif
   97|       |
   98|  42.2k|    if (hdr->reduced_still_picture_header) {
  ------------------
  |  Branch (98:9): [True: 22.2k, False: 19.9k]
  ------------------
   99|  22.2k|        hdr->num_operating_points = 1;
  100|  22.2k|        hdr->operating_points[0].major_level = dav1d_get_bits(gb, 3);
  101|  22.2k|        hdr->operating_points[0].minor_level = dav1d_get_bits(gb, 2);
  102|  22.2k|        hdr->operating_points[0].initial_display_delay = 10;
  103|  22.2k|    } else {
  104|  19.9k|        hdr->timing_info_present = dav1d_get_bit(gb);
  105|  19.9k|        if (hdr->timing_info_present) {
  ------------------
  |  Branch (105:13): [True: 1.29k, False: 18.6k]
  ------------------
  106|  1.29k|            hdr->num_units_in_tick = dav1d_get_bits(gb, 32);
  107|  1.29k|            hdr->time_scale = dav1d_get_bits(gb, 32);
  108|  1.29k|            if (strict_std_compliance && (!hdr->num_units_in_tick || !hdr->time_scale))
  ------------------
  |  Branch (108:17): [True: 0, False: 1.29k]
  |  Branch (108:43): [True: 0, False: 0]
  |  Branch (108:70): [True: 0, False: 0]
  ------------------
  109|      0|                goto error;
  110|  1.29k|            hdr->equal_picture_interval = dav1d_get_bit(gb);
  111|  1.29k|            if (hdr->equal_picture_interval) {
  ------------------
  |  Branch (111:17): [True: 650, False: 640]
  ------------------
  112|    650|                const unsigned num_ticks_per_picture = dav1d_get_vlc(gb);
  113|    650|                if (num_ticks_per_picture == UINT32_MAX)
  ------------------
  |  Branch (113:21): [True: 73, False: 577]
  ------------------
  114|     73|                    goto error;
  115|    577|                hdr->num_ticks_per_picture = num_ticks_per_picture + 1;
  116|    577|            }
  117|       |
  118|  1.21k|            hdr->decoder_model_info_present = dav1d_get_bit(gb);
  119|  1.21k|            if (hdr->decoder_model_info_present) {
  ------------------
  |  Branch (119:17): [True: 853, False: 364]
  ------------------
  120|    853|                hdr->encoder_decoder_buffer_delay_length = dav1d_get_bits(gb, 5) + 1;
  121|    853|                hdr->num_units_in_decoding_tick = dav1d_get_bits(gb, 32);
  122|    853|                if (strict_std_compliance && !hdr->num_units_in_decoding_tick)
  ------------------
  |  Branch (122:21): [True: 0, False: 853]
  |  Branch (122:46): [True: 0, False: 0]
  ------------------
  123|      0|                    goto error;
  124|    853|                hdr->buffer_removal_delay_length = dav1d_get_bits(gb, 5) + 1;
  125|    853|                hdr->frame_presentation_delay_length = dav1d_get_bits(gb, 5) + 1;
  126|    853|            }
  127|  1.21k|        }
  128|       |#if DEBUG_SEQ_HDR
  129|       |        printf("SEQHDR: post-timinginfo: off=%u\n",
  130|       |               dav1d_get_bits_pos(gb) - init_bit_pos);
  131|       |#endif
  132|       |
  133|  19.8k|        hdr->display_model_info_present = dav1d_get_bit(gb);
  134|  19.8k|        hdr->num_operating_points = dav1d_get_bits(gb, 5) + 1;
  135|  54.3k|        for (int i = 0; i < hdr->num_operating_points; i++) {
  ------------------
  |  Branch (135:25): [True: 37.0k, False: 17.2k]
  ------------------
  136|  37.0k|            struct Dav1dSequenceHeaderOperatingPoint *const op =
  137|  37.0k|                &hdr->operating_points[i];
  138|  37.0k|            op->idc = dav1d_get_bits(gb, 12);
  139|  37.0k|            if (op->idc && (!(op->idc & 0xff) || !(op->idc & 0xf00)))
  ------------------
  |  Branch (139:17): [True: 25.1k, False: 11.8k]
  |  Branch (139:29): [True: 2.38k, False: 22.7k]
  |  Branch (139:50): [True: 236, False: 22.5k]
  ------------------
  140|  2.61k|                goto error;
  141|  34.4k|            op->major_level = 2 + dav1d_get_bits(gb, 3);
  142|  34.4k|            op->minor_level = dav1d_get_bits(gb, 2);
  143|  34.4k|            if (op->major_level > 3)
  ------------------
  |  Branch (143:17): [True: 9.81k, False: 24.6k]
  ------------------
  144|  9.81k|                op->tier = dav1d_get_bit(gb);
  145|  34.4k|            if (hdr->decoder_model_info_present) {
  ------------------
  |  Branch (145:17): [True: 3.37k, False: 31.0k]
  ------------------
  146|  3.37k|                op->decoder_model_param_present = dav1d_get_bit(gb);
  147|  3.37k|                if (op->decoder_model_param_present) {
  ------------------
  |  Branch (147:21): [True: 1.34k, False: 2.03k]
  ------------------
  148|  1.34k|                    struct Dav1dSequenceHeaderOperatingParameterInfo *const opi =
  149|  1.34k|                        &hdr->operating_parameter_info[i];
  150|  1.34k|                    opi->decoder_buffer_delay =
  151|  1.34k|                        dav1d_get_bits(gb, hdr->encoder_decoder_buffer_delay_length);
  152|  1.34k|                    opi->encoder_buffer_delay =
  153|  1.34k|                        dav1d_get_bits(gb, hdr->encoder_decoder_buffer_delay_length);
  154|  1.34k|                    opi->low_delay_mode = dav1d_get_bit(gb);
  155|  1.34k|                }
  156|  3.37k|            }
  157|  34.4k|            if (hdr->display_model_info_present)
  ------------------
  |  Branch (157:17): [True: 5.64k, False: 28.7k]
  ------------------
  158|  5.64k|                op->display_model_param_present = dav1d_get_bit(gb);
  159|  34.4k|            op->initial_display_delay =
  160|  34.4k|                op->display_model_param_present ? dav1d_get_bits(gb, 4) + 1 : 10;
  ------------------
  |  Branch (160:17): [True: 1.24k, False: 33.1k]
  ------------------
  161|  34.4k|        }
  162|       |#if DEBUG_SEQ_HDR
  163|       |        printf("SEQHDR: post-operating-points: off=%u\n",
  164|       |               dav1d_get_bits_pos(gb) - init_bit_pos);
  165|       |#endif
  166|  19.8k|    }
  167|       |
  168|  39.5k|    hdr->width_n_bits = dav1d_get_bits(gb, 4) + 1;
  169|  39.5k|    hdr->height_n_bits = dav1d_get_bits(gb, 4) + 1;
  170|  39.5k|    hdr->max_width = dav1d_get_bits(gb, hdr->width_n_bits) + 1;
  171|  39.5k|    hdr->max_height = dav1d_get_bits(gb, hdr->height_n_bits) + 1;
  172|       |#if DEBUG_SEQ_HDR
  173|       |    printf("SEQHDR: post-size: off=%u\n",
  174|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  175|       |#endif
  176|  39.5k|    if (!hdr->reduced_still_picture_header) {
  ------------------
  |  Branch (176:9): [True: 17.2k, False: 22.2k]
  ------------------
  177|  17.2k|        hdr->frame_id_numbers_present = dav1d_get_bit(gb);
  178|  17.2k|        if (hdr->frame_id_numbers_present) {
  ------------------
  |  Branch (178:13): [True: 870, False: 16.4k]
  ------------------
  179|    870|            hdr->delta_frame_id_n_bits = dav1d_get_bits(gb, 4) + 2;
  180|    870|            hdr->frame_id_n_bits = dav1d_get_bits(gb, 3) + hdr->delta_frame_id_n_bits + 1;
  181|    870|        }
  182|  17.2k|    }
  183|       |#if DEBUG_SEQ_HDR
  184|       |    printf("SEQHDR: post-frame-id-numbers-present: off=%u\n",
  185|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  186|       |#endif
  187|       |
  188|  39.5k|    hdr->sb128 = dav1d_get_bit(gb);
  189|  39.5k|    hdr->filter_intra = dav1d_get_bit(gb);
  190|  39.5k|    hdr->intra_edge_filter = dav1d_get_bit(gb);
  191|  39.5k|    if (hdr->reduced_still_picture_header) {
  ------------------
  |  Branch (191:9): [True: 22.2k, False: 17.2k]
  ------------------
  192|  22.2k|        hdr->screen_content_tools = DAV1D_ADAPTIVE;
  193|  22.2k|        hdr->force_integer_mv = DAV1D_ADAPTIVE;
  194|  22.2k|    } else {
  195|  17.2k|        hdr->inter_intra = dav1d_get_bit(gb);
  196|  17.2k|        hdr->masked_compound = dav1d_get_bit(gb);
  197|  17.2k|        hdr->warped_motion = dav1d_get_bit(gb);
  198|  17.2k|        hdr->dual_filter = dav1d_get_bit(gb);
  199|  17.2k|        hdr->order_hint = dav1d_get_bit(gb);
  200|  17.2k|        if (hdr->order_hint) {
  ------------------
  |  Branch (200:13): [True: 10.3k, False: 6.92k]
  ------------------
  201|  10.3k|            hdr->jnt_comp = dav1d_get_bit(gb);
  202|  10.3k|            hdr->ref_frame_mvs = dav1d_get_bit(gb);
  203|  10.3k|        }
  204|  17.2k|        hdr->screen_content_tools = dav1d_get_bit(gb) ? DAV1D_ADAPTIVE : dav1d_get_bit(gb);
  ------------------
  |  Branch (204:37): [True: 13.5k, False: 3.76k]
  ------------------
  205|       |    #if DEBUG_SEQ_HDR
  206|       |        printf("SEQHDR: post-screentools: off=%u\n",
  207|       |               dav1d_get_bits_pos(gb) - init_bit_pos);
  208|       |    #endif
  209|  17.2k|        hdr->force_integer_mv = hdr->screen_content_tools ?
  ------------------
  |  Branch (209:33): [True: 14.7k, False: 2.49k]
  ------------------
  210|  14.7k|                                dav1d_get_bit(gb) ? DAV1D_ADAPTIVE : dav1d_get_bit(gb) : 2;
  ------------------
  |  Branch (210:33): [True: 7.95k, False: 6.82k]
  ------------------
  211|  17.2k|        if (hdr->order_hint)
  ------------------
  |  Branch (211:13): [True: 10.3k, False: 6.92k]
  ------------------
  212|  10.3k|            hdr->order_hint_n_bits = dav1d_get_bits(gb, 3) + 1;
  213|  17.2k|    }
  214|  39.5k|    hdr->super_res = dav1d_get_bit(gb);
  215|  39.5k|    hdr->cdef = dav1d_get_bit(gb);
  216|  39.5k|    hdr->restoration = dav1d_get_bit(gb);
  217|       |#if DEBUG_SEQ_HDR
  218|       |    printf("SEQHDR: post-featurebits: off=%u\n",
  219|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  220|       |#endif
  221|       |
  222|  39.5k|    hdr->hbd = dav1d_get_bit(gb);
  223|  39.5k|    if (hdr->profile == 2 && hdr->hbd)
  ------------------
  |  Branch (223:9): [True: 14.4k, False: 25.0k]
  |  Branch (223:30): [True: 10.9k, False: 3.47k]
  ------------------
  224|  10.9k|        hdr->hbd += dav1d_get_bit(gb);
  225|  39.5k|    if (hdr->profile != 1)
  ------------------
  |  Branch (225:9): [True: 29.5k, False: 10.0k]
  ------------------
  226|  29.5k|        hdr->monochrome = dav1d_get_bit(gb);
  227|  39.5k|    hdr->color_description_present = dav1d_get_bit(gb);
  228|  39.5k|    if (hdr->color_description_present) {
  ------------------
  |  Branch (228:9): [True: 6.02k, False: 33.5k]
  ------------------
  229|  6.02k|        hdr->pri = dav1d_get_bits(gb, 8);
  230|  6.02k|        hdr->trc = dav1d_get_bits(gb, 8);
  231|  6.02k|        hdr->mtrx = dav1d_get_bits(gb, 8);
  232|  33.5k|    } else {
  233|  33.5k|        hdr->pri = DAV1D_COLOR_PRI_UNKNOWN;
  234|  33.5k|        hdr->trc = DAV1D_TRC_UNKNOWN;
  235|  33.5k|        hdr->mtrx = DAV1D_MC_UNKNOWN;
  236|  33.5k|    }
  237|  39.5k|    if (hdr->monochrome) {
  ------------------
  |  Branch (237:9): [True: 9.26k, False: 30.2k]
  ------------------
  238|  9.26k|        hdr->color_range = dav1d_get_bit(gb);
  239|  9.26k|        hdr->layout = DAV1D_PIXEL_LAYOUT_I400;
  240|  9.26k|        hdr->ss_hor = hdr->ss_ver = 1;
  241|  9.26k|        hdr->chr = DAV1D_CHR_UNKNOWN;
  242|  30.2k|    } else if (hdr->pri == DAV1D_COLOR_PRI_BT709 &&
  ------------------
  |  Branch (242:16): [True: 2.12k, False: 28.1k]
  ------------------
  243|  2.12k|               hdr->trc == DAV1D_TRC_SRGB &&
  ------------------
  |  Branch (243:16): [True: 1.77k, False: 356]
  ------------------
  244|  1.77k|               hdr->mtrx == DAV1D_MC_IDENTITY)
  ------------------
  |  Branch (244:16): [True: 1.39k, False: 373]
  ------------------
  245|  1.39k|    {
  246|  1.39k|        hdr->layout = DAV1D_PIXEL_LAYOUT_I444;
  247|  1.39k|        hdr->color_range = 1;
  248|  1.39k|        if (hdr->profile != 1 && !(hdr->profile == 2 && hdr->hbd == 2))
  ------------------
  |  Branch (248:13): [True: 1.08k, False: 315]
  |  Branch (248:36): [True: 198, False: 885]
  |  Branch (248:57): [True: 93, False: 105]
  ------------------
  249|    990|            goto error;
  250|  28.8k|    } else {
  251|  28.8k|        hdr->color_range = dav1d_get_bit(gb);
  252|  28.8k|        switch (hdr->profile) {
  ------------------
  |  Branch (252:17): [True: 28.8k, False: 0]
  ------------------
  253|  11.2k|        case 0: hdr->layout = DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (253:9): [True: 11.2k, False: 17.6k]
  ------------------
  254|  11.2k|                hdr->ss_hor = hdr->ss_ver = 1;
  255|  11.2k|                break;
  256|  9.73k|        case 1: hdr->layout = DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (256:9): [True: 9.73k, False: 19.1k]
  ------------------
  257|  9.73k|                break;
  258|  7.89k|        case 2:
  ------------------
  |  Branch (258:9): [True: 7.89k, False: 21.0k]
  ------------------
  259|  7.89k|            if (hdr->hbd == 2) {
  ------------------
  |  Branch (259:17): [True: 5.20k, False: 2.69k]
  ------------------
  260|  5.20k|                hdr->ss_hor = dav1d_get_bit(gb);
  261|  5.20k|                if (hdr->ss_hor)
  ------------------
  |  Branch (261:21): [True: 1.37k, False: 3.82k]
  ------------------
  262|  1.37k|                    hdr->ss_ver = dav1d_get_bit(gb);
  263|  5.20k|            } else
  264|  2.69k|                hdr->ss_hor = 1;
  265|  7.89k|            hdr->layout = hdr->ss_hor ?
  ------------------
  |  Branch (265:27): [True: 4.06k, False: 3.82k]
  ------------------
  266|  4.06k|                          hdr->ss_ver ? DAV1D_PIXEL_LAYOUT_I420 :
  ------------------
  |  Branch (266:27): [True: 1.28k, False: 2.78k]
  ------------------
  267|  4.06k|                                        DAV1D_PIXEL_LAYOUT_I422 :
  268|  7.89k|                                        DAV1D_PIXEL_LAYOUT_I444;
  269|  7.89k|            break;
  270|  28.8k|        }
  271|  28.8k|        hdr->chr = (hdr->ss_hor & hdr->ss_ver) ?
  ------------------
  |  Branch (271:20): [True: 12.5k, False: 16.3k]
  ------------------
  272|  16.3k|                   dav1d_get_bits(gb, 2) : DAV1D_CHR_UNKNOWN;
  273|  28.8k|    }
  274|  38.5k|    if (strict_std_compliance &&
  ------------------
  |  Branch (274:9): [True: 0, False: 38.5k]
  ------------------
  275|      0|        hdr->mtrx == DAV1D_MC_IDENTITY && hdr->layout != DAV1D_PIXEL_LAYOUT_I444)
  ------------------
  |  Branch (275:9): [True: 0, False: 0]
  |  Branch (275:43): [True: 0, False: 0]
  ------------------
  276|      0|    {
  277|      0|        goto error;
  278|      0|    }
  279|  38.5k|    if (!hdr->monochrome)
  ------------------
  |  Branch (279:9): [True: 29.3k, False: 9.26k]
  ------------------
  280|  29.3k|        hdr->separate_uv_delta_q = dav1d_get_bit(gb);
  281|       |#if DEBUG_SEQ_HDR
  282|       |    printf("SEQHDR: post-colorinfo: off=%u\n",
  283|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  284|       |#endif
  285|       |
  286|  38.5k|    hdr->film_grain_present = dav1d_get_bit(gb);
  287|       |#if DEBUG_SEQ_HDR
  288|       |    printf("SEQHDR: post-filmgrain: off=%u\n",
  289|       |           dav1d_get_bits_pos(gb) - init_bit_pos);
  290|       |#endif
  291|       |
  292|       |    // We needn't bother flushing the OBU here: we'll check we didn't
  293|       |    // overrun in the caller and will then discard gb, so there's no
  294|       |    // point in setting its position properly.
  295|       |
  296|  38.5k|    return check_trailing_bits(gb, strict_std_compliance);
  297|       |
  298|  5.49k|error:
  299|  5.49k|    return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  5.49k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  300|  38.5k|}
obu.c:parse_frame_hdr:
  409|   417k|static int parse_frame_hdr(Dav1dContext *const c, GetBits *const gb) {
  410|   417k|#define DEBUG_FRAME_HDR 0
  411|       |
  412|       |#if DEBUG_FRAME_HDR
  413|       |    const uint8_t *const init_ptr = gb->ptr;
  414|       |#endif
  415|   417k|    const Dav1dSequenceHeader *const seqhdr = c->seq_hdr;
  416|   417k|    Dav1dFrameHeader *const hdr = c->frame_hdr;
  417|       |
  418|   417k|    if (!seqhdr->reduced_still_picture_header)
  ------------------
  |  Branch (418:9): [True: 166k, False: 250k]
  ------------------
  419|   166k|        hdr->show_existing_frame = dav1d_get_bit(gb);
  420|       |#if DEBUG_FRAME_HDR
  421|       |    printf("HDR: post-show_existing_frame: off=%td\n",
  422|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  423|       |#endif
  424|   417k|    if (hdr->show_existing_frame) {
  ------------------
  |  Branch (424:9): [True: 32.8k, False: 384k]
  ------------------
  425|  32.8k|        hdr->existing_frame_idx = dav1d_get_bits(gb, 3);
  426|  32.8k|        if (seqhdr->decoder_model_info_present && !seqhdr->equal_picture_interval)
  ------------------
  |  Branch (426:13): [True: 604, False: 32.1k]
  |  Branch (426:51): [True: 381, False: 223]
  ------------------
  427|    381|            hdr->frame_presentation_delay = dav1d_get_bits(gb, seqhdr->frame_presentation_delay_length);
  428|  32.8k|        if (seqhdr->frame_id_numbers_present) {
  ------------------
  |  Branch (428:13): [True: 962, False: 31.8k]
  ------------------
  429|    962|            hdr->frame_id = dav1d_get_bits(gb, seqhdr->frame_id_n_bits);
  430|    962|            Dav1dFrameHeader *const ref_frame_hdr = c->refs[hdr->existing_frame_idx].p.p.frame_hdr;
  431|    962|            if (!ref_frame_hdr || ref_frame_hdr->frame_id != hdr->frame_id) goto error;
  ------------------
  |  Branch (431:17): [True: 400, False: 562]
  |  Branch (431:35): [True: 162, False: 400]
  ------------------
  432|    962|        }
  433|  32.2k|        return 0;
  434|  32.8k|    }
  435|       |
  436|   384k|    if (seqhdr->reduced_still_picture_header) {
  ------------------
  |  Branch (436:9): [True: 250k, False: 133k]
  ------------------
  437|   250k|        hdr->frame_type = DAV1D_FRAME_TYPE_KEY;
  438|   250k|        hdr->show_frame = 1;
  439|   250k|    } else {
  440|   133k|        hdr->frame_type = dav1d_get_bits(gb, 2);
  441|   133k|        hdr->show_frame = dav1d_get_bit(gb);
  442|   133k|    }
  443|   384k|    if (hdr->show_frame) {
  ------------------
  |  Branch (443:9): [True: 312k, False: 72.6k]
  ------------------
  444|   312k|        if (seqhdr->decoder_model_info_present && !seqhdr->equal_picture_interval)
  ------------------
  |  Branch (444:13): [True: 9.83k, False: 302k]
  |  Branch (444:51): [True: 9.25k, False: 580]
  ------------------
  445|  9.25k|            hdr->frame_presentation_delay = dav1d_get_bits(gb, seqhdr->frame_presentation_delay_length);
  446|   312k|        hdr->showable_frame = hdr->frame_type != DAV1D_FRAME_TYPE_KEY;
  447|   312k|    } else
  448|  72.6k|        hdr->showable_frame = dav1d_get_bit(gb);
  449|   384k|    hdr->error_resilient_mode =
  450|   384k|        (hdr->frame_type == DAV1D_FRAME_TYPE_KEY && hdr->show_frame) ||
  ------------------
  |  Branch (450:10): [True: 268k, False: 116k]
  |  Branch (450:53): [True: 265k, False: 3.09k]
  ------------------
  451|   119k|        hdr->frame_type == DAV1D_FRAME_TYPE_SWITCH ||
  ------------------
  |  Branch (451:9): [True: 2.86k, False: 116k]
  ------------------
  452|   116k|        seqhdr->reduced_still_picture_header || dav1d_get_bit(gb);
  ------------------
  |  Branch (452:9): [True: 0, False: 116k]
  |  Branch (452:49): [True: 2.27k, False: 113k]
  ------------------
  453|       |#if DEBUG_FRAME_HDR
  454|       |    printf("HDR: post-frametype_bits: off=%td\n",
  455|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  456|       |#endif
  457|   384k|    hdr->disable_cdf_update = dav1d_get_bit(gb);
  458|   384k|    hdr->allow_screen_content_tools = seqhdr->screen_content_tools == DAV1D_ADAPTIVE ?
  ------------------
  |  Branch (458:39): [True: 352k, False: 32.2k]
  ------------------
  459|   352k|                                      dav1d_get_bit(gb) : seqhdr->screen_content_tools;
  460|   384k|    if (hdr->allow_screen_content_tools)
  ------------------
  |  Branch (460:9): [True: 293k, False: 91.6k]
  ------------------
  461|   293k|        hdr->force_integer_mv = seqhdr->force_integer_mv == DAV1D_ADAPTIVE ?
  ------------------
  |  Branch (461:33): [True: 240k, False: 52.9k]
  ------------------
  462|   240k|                                dav1d_get_bit(gb) : seqhdr->force_integer_mv;
  463|       |
  464|   384k|    if (IS_KEY_OR_INTRA(hdr))
  ------------------
  |  |   43|   384k|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|   384k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 270k, False: 114k]
  |  |  ------------------
  ------------------
  465|   270k|        hdr->force_integer_mv = 1;
  466|       |
  467|   384k|    if (seqhdr->frame_id_numbers_present)
  ------------------
  |  Branch (467:9): [True: 3.38k, False: 381k]
  ------------------
  468|  3.38k|        hdr->frame_id = dav1d_get_bits(gb, seqhdr->frame_id_n_bits);
  469|       |
  470|   384k|    if (!seqhdr->reduced_still_picture_header)
  ------------------
  |  Branch (470:9): [True: 133k, False: 250k]
  ------------------
  471|   133k|        hdr->frame_size_override = hdr->frame_type == DAV1D_FRAME_TYPE_SWITCH ? 1 : dav1d_get_bit(gb);
  ------------------
  |  Branch (471:36): [True: 2.86k, False: 131k]
  ------------------
  472|       |#if DEBUG_FRAME_HDR
  473|       |    printf("HDR: post-frame_size_override_flag: off=%td\n",
  474|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  475|       |#endif
  476|   384k|    if (seqhdr->order_hint)
  ------------------
  |  Branch (476:9): [True: 120k, False: 264k]
  ------------------
  477|   120k|        hdr->frame_offset = dav1d_get_bits(gb, seqhdr->order_hint_n_bits);
  478|   384k|    hdr->primary_ref_frame = !hdr->error_resilient_mode && IS_INTER_OR_SWITCH(hdr) ?
  ------------------
  |  |   36|   113k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 111k, False: 2.48k]
  |  |  ------------------
  ------------------
  |  Branch (478:30): [True: 113k, False: 270k]
  ------------------
  479|   273k|                             dav1d_get_bits(gb, 3) : DAV1D_PRIMARY_REF_NONE;
  ------------------
  |  |   45|   658k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  480|       |
  481|   384k|    if (seqhdr->decoder_model_info_present) {
  ------------------
  |  Branch (481:9): [True: 10.4k, False: 374k]
  ------------------
  482|  10.4k|        hdr->buffer_removal_time_present = dav1d_get_bit(gb);
  483|  10.4k|        if (hdr->buffer_removal_time_present) {
  ------------------
  |  Branch (483:13): [True: 3.24k, False: 7.17k]
  ------------------
  484|  9.29k|            for (int i = 0; i < c->seq_hdr->num_operating_points; i++) {
  ------------------
  |  Branch (484:29): [True: 6.05k, False: 3.24k]
  ------------------
  485|  6.05k|                const struct Dav1dSequenceHeaderOperatingPoint *const seqop = &seqhdr->operating_points[i];
  486|  6.05k|                struct Dav1dFrameHeaderOperatingPoint *const op = &hdr->operating_points[i];
  487|  6.05k|                if (seqop->decoder_model_param_present) {
  ------------------
  |  Branch (487:21): [True: 2.01k, False: 4.03k]
  ------------------
  488|  2.01k|                    int in_temporal_layer = (seqop->idc >> hdr->temporal_id) & 1;
  489|  2.01k|                    int in_spatial_layer  = (seqop->idc >> (hdr->spatial_id + 8)) & 1;
  490|  2.01k|                    if (!seqop->idc || (in_temporal_layer && in_spatial_layer))
  ------------------
  |  Branch (490:25): [True: 278, False: 1.73k]
  |  Branch (490:41): [True: 782, False: 957]
  |  Branch (490:62): [True: 513, False: 269]
  ------------------
  491|    791|                        op->buffer_removal_time = dav1d_get_bits(gb, seqhdr->buffer_removal_delay_length);
  492|  2.01k|                }
  493|  6.05k|            }
  494|  3.24k|        }
  495|  10.4k|    }
  496|       |
  497|   384k|    if (IS_KEY_OR_INTRA(hdr)) {
  ------------------
  |  |   43|   384k|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|   384k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 270k, False: 114k]
  |  |  ------------------
  ------------------
  498|   270k|        hdr->refresh_frame_flags = (hdr->frame_type == DAV1D_FRAME_TYPE_KEY &&
  ------------------
  |  Branch (498:37): [True: 268k, False: 1.48k]
  ------------------
  499|   268k|                                    hdr->show_frame) ? 0xff : dav1d_get_bits(gb, 8);
  ------------------
  |  Branch (499:37): [True: 265k, False: 3.09k]
  ------------------
  500|   270k|        if (hdr->refresh_frame_flags != 0xff && hdr->error_resilient_mode && seqhdr->order_hint)
  ------------------
  |  Branch (500:13): [True: 4.52k, False: 265k]
  |  Branch (500:49): [True: 2.09k, False: 2.43k]
  |  Branch (500:78): [True: 1.19k, False: 896]
  ------------------
  501|  10.7k|            for (int i = 0; i < 8; i++)
  ------------------
  |  Branch (501:29): [True: 9.57k, False: 1.19k]
  ------------------
  502|  9.57k|                dav1d_get_bits(gb, seqhdr->order_hint_n_bits);
  503|   270k|        if (c->strict_std_compliance &&
  ------------------
  |  Branch (503:13): [True: 0, False: 270k]
  ------------------
  504|      0|            hdr->frame_type == DAV1D_FRAME_TYPE_INTRA && hdr->refresh_frame_flags == 0xff)
  ------------------
  |  Branch (504:13): [True: 0, False: 0]
  |  Branch (504:58): [True: 0, False: 0]
  ------------------
  505|      0|        {
  506|      0|            goto error;
  507|      0|        }
  508|   270k|        if (read_frame_size(c, gb, 0) < 0) goto error;
  ------------------
  |  Branch (508:13): [True: 0, False: 270k]
  ------------------
  509|   270k|        if (hdr->allow_screen_content_tools && !hdr->super_res.enabled)
  ------------------
  |  Branch (509:13): [True: 239k, False: 30.4k]
  |  Branch (509:48): [True: 237k, False: 2.24k]
  ------------------
  510|   237k|            hdr->allow_intrabc = dav1d_get_bit(gb);
  511|   270k|    } else {
  512|   114k|        hdr->refresh_frame_flags = hdr->frame_type == DAV1D_FRAME_TYPE_SWITCH ? 0xff :
  ------------------
  |  Branch (512:36): [True: 2.86k, False: 111k]
  ------------------
  513|   114k|                                   dav1d_get_bits(gb, 8);
  514|   114k|        if (hdr->error_resilient_mode && seqhdr->order_hint)
  ------------------
  |  Branch (514:13): [True: 3.05k, False: 111k]
  |  Branch (514:42): [True: 2.40k, False: 646]
  ------------------
  515|  21.6k|            for (int i = 0; i < 8; i++)
  ------------------
  |  Branch (515:29): [True: 19.2k, False: 2.40k]
  ------------------
  516|  19.2k|                dav1d_get_bits(gb, seqhdr->order_hint_n_bits);
  517|   114k|        if (seqhdr->order_hint) {
  ------------------
  |  Branch (517:13): [True: 107k, False: 7.14k]
  ------------------
  518|   107k|            hdr->frame_ref_short_signaling = dav1d_get_bit(gb);
  519|   107k|            if (hdr->frame_ref_short_signaling) {
  ------------------
  |  Branch (519:17): [True: 68.7k, False: 38.6k]
  ------------------
  520|  68.7k|                hdr->refidx[0] = dav1d_get_bits(gb, 3);
  521|  68.7k|                hdr->refidx[1] = hdr->refidx[2] = -1;
  522|  68.7k|                hdr->refidx[3] = dav1d_get_bits(gb, 3);
  523|       |
  524|       |                /* +1 allows for unconditional stores, as unused
  525|       |                 * values can be dumped into frame_offset[-1]. */
  526|  68.7k|                int frame_offset_mem[8+1];
  527|  68.7k|                int *const frame_offset = &frame_offset_mem[1];
  528|  68.7k|                int earliest_ref = -1;
  529|   614k|                for (int i = 0, earliest_offset = INT_MAX; i < 8; i++) {
  ------------------
  |  Branch (529:60): [True: 546k, False: 68.1k]
  ------------------
  530|   546k|                    const Dav1dFrameHeader *const refhdr = c->refs[i].p.p.frame_hdr;
  531|   546k|                    if (!refhdr) goto error;
  ------------------
  |  Branch (531:25): [True: 535, False: 545k]
  ------------------
  532|   545k|                    const int diff = get_poc_diff(seqhdr->order_hint_n_bits,
  533|   545k|                                                  refhdr->frame_offset,
  534|   545k|                                                  hdr->frame_offset);
  535|   545k|                    frame_offset[i] = diff;
  536|   545k|                    if (diff < earliest_offset) {
  ------------------
  |  Branch (536:25): [True: 103k, False: 441k]
  ------------------
  537|   103k|                        earliest_offset = diff;
  538|   103k|                        earliest_ref = i;
  539|   103k|                    }
  540|   545k|                }
  541|  68.1k|                frame_offset[hdr->refidx[0]] = INT_MIN; // = reference frame is used
  542|  68.1k|                frame_offset[hdr->refidx[3]] = INT_MIN;
  543|  68.1k|                assert(earliest_ref >= 0);
  ------------------
  |  Branch (543:17): [True: 68.1k, False: 0]
  ------------------
  544|       |
  545|  68.1k|                int refidx = -1;
  546|   613k|                for (int i = 0, latest_offset = 0; i < 8; i++) {
  ------------------
  |  Branch (546:52): [True: 545k, False: 68.1k]
  ------------------
  547|   545k|                    const int hint = frame_offset[i];
  548|   545k|                    if (hint >= latest_offset) {
  ------------------
  |  Branch (548:25): [True: 255k, False: 289k]
  ------------------
  549|   255k|                        latest_offset = hint;
  550|   255k|                        refidx = i;
  551|   255k|                    }
  552|   545k|                }
  553|  68.1k|                frame_offset[refidx] = INT_MIN;
  554|  68.1k|                hdr->refidx[6] = refidx;
  555|       |
  556|   204k|                for (int i = 4; i < 6; i++) {
  ------------------
  |  Branch (556:33): [True: 136k, False: 68.1k]
  ------------------
  557|       |                    /* Unsigned compares to handle negative values. */
  558|   136k|                    unsigned earliest_offset = UINT8_MAX;
  559|   136k|                    refidx = -1;
  560|  1.22M|                    for (int j = 0; j < 8; j++) {
  ------------------
  |  Branch (560:37): [True: 1.09M, False: 136k]
  ------------------
  561|  1.09M|                        const unsigned hint = frame_offset[j];
  562|  1.09M|                        if (hint < earliest_offset) {
  ------------------
  |  Branch (562:29): [True: 128k, False: 961k]
  ------------------
  563|   128k|                            earliest_offset = hint;
  564|   128k|                            refidx = j;
  565|   128k|                        }
  566|  1.09M|                    }
  567|   136k|                    frame_offset[refidx] = INT_MIN;
  568|   136k|                    hdr->refidx[i] = refidx;
  569|   136k|                }
  570|       |
  571|   477k|                for (int i = 1; i < 7; i++) {
  ------------------
  |  Branch (571:33): [True: 409k, False: 68.1k]
  ------------------
  572|   409k|                    refidx = hdr->refidx[i];
  573|   409k|                    if (refidx < 0) {
  ------------------
  |  Branch (573:25): [True: 159k, False: 249k]
  ------------------
  574|   159k|                        unsigned latest_offset = ~UINT8_MAX;
  575|  1.43M|                        for (int j = 0; j < 8; j++) {
  ------------------
  |  Branch (575:41): [True: 1.27M, False: 159k]
  ------------------
  576|  1.27M|                            const unsigned hint = frame_offset[j];
  577|  1.27M|                            if (hint >= latest_offset) {
  ------------------
  |  Branch (577:33): [True: 282k, False: 995k]
  ------------------
  578|   282k|                                latest_offset = hint;
  579|   282k|                                refidx = j;
  580|   282k|                            }
  581|  1.27M|                        }
  582|   159k|                        frame_offset[refidx] = INT_MIN;
  583|   159k|                        hdr->refidx[i] = refidx >= 0 ? refidx : earliest_ref;
  ------------------
  |  Branch (583:42): [True: 109k, False: 49.9k]
  ------------------
  584|   159k|                    }
  585|   409k|                }
  586|  68.1k|            }
  587|   107k|        }
  588|   895k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (588:25): [True: 783k, False: 111k]
  ------------------
  589|   783k|            if (!hdr->frame_ref_short_signaling)
  ------------------
  |  Branch (589:17): [True: 307k, False: 476k]
  ------------------
  590|   307k|                hdr->refidx[i] = dav1d_get_bits(gb, 3);
  591|   783k|            if (seqhdr->frame_id_numbers_present) {
  ------------------
  |  Branch (591:17): [True: 2.43k, False: 781k]
  ------------------
  592|  2.43k|                const unsigned delta_ref_frame_id = dav1d_get_bits(gb, seqhdr->delta_frame_id_n_bits) + 1;
  593|  2.43k|                const unsigned ref_frame_id = (hdr->frame_id + (1 << seqhdr->frame_id_n_bits) - delta_ref_frame_id) & ((1 << seqhdr->frame_id_n_bits) - 1);
  594|  2.43k|                Dav1dFrameHeader *const ref_frame_hdr = c->refs[hdr->refidx[i]].p.p.frame_hdr;
  595|  2.43k|                if (!ref_frame_hdr || ref_frame_hdr->frame_id != ref_frame_id) goto error;
  ------------------
  |  Branch (595:21): [True: 477, False: 1.96k]
  |  Branch (595:39): [True: 1.89k, False: 69]
  ------------------
  596|  2.43k|            }
  597|   783k|        }
  598|   111k|        const int use_ref = !hdr->error_resilient_mode &&
  ------------------
  |  Branch (598:29): [True: 109k, False: 2.28k]
  ------------------
  599|   109k|                            hdr->frame_size_override;
  ------------------
  |  Branch (599:29): [True: 62.0k, False: 47.2k]
  ------------------
  600|   111k|        if (read_frame_size(c, gb, use_ref) < 0) goto error;
  ------------------
  |  Branch (600:13): [True: 259, False: 111k]
  ------------------
  601|   111k|        if (!hdr->force_integer_mv)
  ------------------
  |  Branch (601:13): [True: 70.5k, False: 40.8k]
  ------------------
  602|  70.5k|            hdr->hp = dav1d_get_bit(gb);
  603|   111k|        hdr->subpel_filter_mode = dav1d_get_bit(gb) ? DAV1D_FILTER_SWITCHABLE :
  ------------------
  |  Branch (603:35): [True: 64.4k, False: 46.9k]
  ------------------
  604|   111k|                                                      dav1d_get_bits(gb, 2);
  605|   111k|        hdr->switchable_motion_mode = dav1d_get_bit(gb);
  606|   111k|        if (!hdr->error_resilient_mode && seqhdr->ref_frame_mvs &&
  ------------------
  |  Branch (606:13): [True: 109k, False: 2.28k]
  |  Branch (606:43): [True: 91.9k, False: 17.1k]
  ------------------
  607|  91.9k|            seqhdr->order_hint && IS_INTER_OR_SWITCH(hdr))
  ------------------
  |  |   36|  91.9k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 91.9k, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (607:13): [True: 91.9k, False: 0]
  ------------------
  608|  91.9k|        {
  609|  91.9k|            hdr->use_ref_frame_mvs = dav1d_get_bit(gb);
  610|  91.9k|        }
  611|   111k|    }
  612|       |#if DEBUG_FRAME_HDR
  613|       |    printf("HDR: post-frametype-specific-bits: off=%td\n",
  614|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  615|       |#endif
  616|       |
  617|   381k|    if (!seqhdr->reduced_still_picture_header && !hdr->disable_cdf_update)
  ------------------
  |  Branch (617:9): [True: 130k, False: 250k]
  |  Branch (617:50): [True: 122k, False: 7.86k]
  ------------------
  618|   122k|        hdr->refresh_context = !dav1d_get_bit(gb);
  619|       |#if DEBUG_FRAME_HDR
  620|       |    printf("HDR: post-refresh_context: off=%td\n",
  621|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  622|       |#endif
  623|       |
  624|       |    // tile data
  625|   381k|    hdr->tiling.uniform = dav1d_get_bit(gb);
  626|   381k|    const int sbsz_min1 = (64 << seqhdr->sb128) - 1;
  627|   381k|    const int sbsz_log2 = 6 + seqhdr->sb128;
  628|   381k|    const int sbw = (hdr->width[0] + sbsz_min1) >> sbsz_log2;
  629|   381k|    const int sbh = (hdr->height + sbsz_min1) >> sbsz_log2;
  630|   381k|    const int max_tile_width_sb = 4096 >> sbsz_log2;
  631|   381k|    const int max_tile_area_sb = 4096 * 2304 >> (2 * sbsz_log2);
  632|   381k|    hdr->tiling.min_log2_cols = tile_log2(max_tile_width_sb, sbw);
  633|   381k|    hdr->tiling.max_log2_cols = tile_log2(1, imin(sbw, DAV1D_MAX_TILE_COLS));
  ------------------
  |  |   41|   381k|#define DAV1D_MAX_TILE_COLS 64
  ------------------
  634|   381k|    hdr->tiling.max_log2_rows = tile_log2(1, imin(sbh, DAV1D_MAX_TILE_ROWS));
  ------------------
  |  |   42|   381k|#define DAV1D_MAX_TILE_ROWS 64
  ------------------
  635|   381k|    const int min_log2_tiles = imax(tile_log2(max_tile_area_sb, sbw * sbh),
  636|   381k|                              hdr->tiling.min_log2_cols);
  637|   381k|    if (hdr->tiling.uniform) {
  ------------------
  |  Branch (637:9): [True: 326k, False: 54.7k]
  ------------------
  638|   326k|        for (hdr->tiling.log2_cols = hdr->tiling.min_log2_cols;
  639|   333k|             hdr->tiling.log2_cols < hdr->tiling.max_log2_cols && dav1d_get_bit(gb);
  ------------------
  |  Branch (639:14): [True: 92.6k, False: 240k]
  |  Branch (639:67): [True: 6.69k, False: 85.9k]
  ------------------
  640|   326k|             hdr->tiling.log2_cols++) ;
  641|   326k|        const int tile_w = 1 + ((sbw - 1) >> hdr->tiling.log2_cols);
  642|   326k|        hdr->tiling.cols = 0;
  643|   676k|        for (int sbx = 0; sbx < sbw; sbx += tile_w, hdr->tiling.cols++)
  ------------------
  |  Branch (643:27): [True: 350k, False: 326k]
  ------------------
  644|   350k|            hdr->tiling.col_start_sb[hdr->tiling.cols] = sbx;
  645|   326k|        hdr->tiling.min_log2_rows =
  646|   326k|            imax(min_log2_tiles - hdr->tiling.log2_cols, 0);
  647|       |
  648|   326k|        for (hdr->tiling.log2_rows = hdr->tiling.min_log2_rows;
  649|   347k|             hdr->tiling.log2_rows < hdr->tiling.max_log2_rows && dav1d_get_bit(gb);
  ------------------
  |  Branch (649:14): [True: 113k, False: 233k]
  |  Branch (649:67): [True: 20.2k, False: 93.3k]
  ------------------
  650|   326k|             hdr->tiling.log2_rows++) ;
  651|   326k|        const int tile_h = 1 + ((sbh - 1) >> hdr->tiling.log2_rows);
  652|   326k|        hdr->tiling.rows = 0;
  653|   720k|        for (int sby = 0; sby < sbh; sby += tile_h, hdr->tiling.rows++)
  ------------------
  |  Branch (653:27): [True: 393k, False: 326k]
  ------------------
  654|   393k|            hdr->tiling.row_start_sb[hdr->tiling.rows] = sby;
  655|   326k|    } else {
  656|  54.7k|        hdr->tiling.cols = 0;
  657|  54.7k|        int widest_tile = 0, max_tile_area_sb = sbw * sbh;
  658|   204k|        for (int sbx = 0; sbx < sbw && hdr->tiling.cols < DAV1D_MAX_TILE_COLS; hdr->tiling.cols++) {
  ------------------
  |  |   41|   150k|#define DAV1D_MAX_TILE_COLS 64
  ------------------
  |  Branch (658:27): [True: 150k, False: 54.4k]
  |  Branch (658:40): [True: 149k, False: 320]
  ------------------
  659|   149k|            const int tile_width_sb = imin(sbw - sbx, max_tile_width_sb);
  660|   149k|            const int tile_w = (tile_width_sb > 1) ? 1 + dav1d_get_uniform(gb, tile_width_sb) : 1;
  ------------------
  |  Branch (660:32): [True: 117k, False: 32.3k]
  ------------------
  661|   149k|            hdr->tiling.col_start_sb[hdr->tiling.cols] = sbx;
  662|   149k|            sbx += tile_w;
  663|   149k|            widest_tile = imax(widest_tile, tile_w);
  664|   149k|        }
  665|  54.7k|        hdr->tiling.log2_cols = tile_log2(1, hdr->tiling.cols);
  666|  54.7k|        if (min_log2_tiles) max_tile_area_sb >>= min_log2_tiles + 1;
  ------------------
  |  Branch (666:13): [True: 1.12k, False: 53.6k]
  ------------------
  667|  54.7k|        const int max_tile_height_sb = imax(max_tile_area_sb / widest_tile, 1);
  668|       |
  669|  54.7k|        hdr->tiling.rows = 0;
  670|   151k|        for (int sby = 0; sby < sbh && hdr->tiling.rows < DAV1D_MAX_TILE_ROWS; hdr->tiling.rows++) {
  ------------------
  |  |   42|  97.0k|#define DAV1D_MAX_TILE_ROWS 64
  ------------------
  |  Branch (670:27): [True: 97.0k, False: 54.6k]
  |  Branch (670:40): [True: 96.9k, False: 144]
  ------------------
  671|  96.9k|            const int tile_height_sb = imin(sbh - sby, max_tile_height_sb);
  672|  96.9k|            const int tile_h = (tile_height_sb > 1) ? 1 + dav1d_get_uniform(gb, tile_height_sb) : 1;
  ------------------
  |  Branch (672:32): [True: 52.8k, False: 44.0k]
  ------------------
  673|  96.9k|            hdr->tiling.row_start_sb[hdr->tiling.rows] = sby;
  674|  96.9k|            sby += tile_h;
  675|  96.9k|        }
  676|  54.7k|        hdr->tiling.log2_rows = tile_log2(1, hdr->tiling.rows);
  677|  54.7k|    }
  678|   381k|    hdr->tiling.col_start_sb[hdr->tiling.cols] = sbw;
  679|   381k|    hdr->tiling.row_start_sb[hdr->tiling.rows] = sbh;
  680|   381k|    if (hdr->tiling.log2_cols || hdr->tiling.log2_rows) {
  ------------------
  |  Branch (680:9): [True: 14.4k, False: 367k]
  |  Branch (680:34): [True: 21.6k, False: 345k]
  ------------------
  681|  36.1k|        hdr->tiling.update = dav1d_get_bits(gb, hdr->tiling.log2_cols + hdr->tiling.log2_rows);
  682|  36.1k|        if (hdr->tiling.update >= hdr->tiling.cols * hdr->tiling.rows)
  ------------------
  |  Branch (682:13): [True: 417, False: 35.6k]
  ------------------
  683|    417|            goto error;
  684|  35.6k|        hdr->tiling.n_bytes = dav1d_get_bits(gb, 2) + 1;
  685|  35.6k|    }
  686|       |#if DEBUG_FRAME_HDR
  687|       |    printf("HDR: post-tiling: off=%td\n",
  688|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  689|       |#endif
  690|       |
  691|       |    // quant data
  692|   381k|    hdr->quant.yac = dav1d_get_bits(gb, 8);
  693|   381k|    if (dav1d_get_bit(gb))
  ------------------
  |  Branch (693:9): [True: 31.1k, False: 350k]
  ------------------
  694|  31.1k|        hdr->quant.ydc_delta = dav1d_get_sbits(gb, 7);
  695|   381k|    if (!seqhdr->monochrome) {
  ------------------
  |  Branch (695:9): [True: 304k, False: 76.9k]
  ------------------
  696|       |        // If the sequence header says that delta_q might be different
  697|       |        // for U, V, we must check whether it actually is for this
  698|       |        // frame.
  699|   304k|        const int diff_uv_delta = seqhdr->separate_uv_delta_q ? dav1d_get_bit(gb) : 0;
  ------------------
  |  Branch (699:35): [True: 26.4k, False: 277k]
  ------------------
  700|   304k|        if (dav1d_get_bit(gb))
  ------------------
  |  Branch (700:13): [True: 16.2k, False: 288k]
  ------------------
  701|  16.2k|            hdr->quant.udc_delta = dav1d_get_sbits(gb, 7);
  702|   304k|        if (dav1d_get_bit(gb))
  ------------------
  |  Branch (702:13): [True: 15.6k, False: 288k]
  ------------------
  703|  15.6k|            hdr->quant.uac_delta = dav1d_get_sbits(gb, 7);
  704|   304k|        if (diff_uv_delta) {
  ------------------
  |  Branch (704:13): [True: 5.53k, False: 298k]
  ------------------
  705|  5.53k|            if (dav1d_get_bit(gb))
  ------------------
  |  Branch (705:17): [True: 1.26k, False: 4.27k]
  ------------------
  706|  1.26k|                hdr->quant.vdc_delta = dav1d_get_sbits(gb, 7);
  707|  5.53k|            if (dav1d_get_bit(gb))
  ------------------
  |  Branch (707:17): [True: 1.71k, False: 3.82k]
  ------------------
  708|  1.71k|                hdr->quant.vac_delta = dav1d_get_sbits(gb, 7);
  709|   298k|        } else {
  710|   298k|            hdr->quant.vdc_delta = hdr->quant.udc_delta;
  711|   298k|            hdr->quant.vac_delta = hdr->quant.uac_delta;
  712|   298k|        }
  713|   304k|    }
  714|       |#if DEBUG_FRAME_HDR
  715|       |    printf("HDR: post-quant: off=%td\n",
  716|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  717|       |#endif
  718|   381k|    hdr->quant.qm = dav1d_get_bit(gb);
  719|   381k|    if (hdr->quant.qm) {
  ------------------
  |  Branch (719:9): [True: 28.6k, False: 352k]
  ------------------
  720|  28.6k|        hdr->quant.qm_y = dav1d_get_bits(gb, 4);
  721|  28.6k|        hdr->quant.qm_u = dav1d_get_bits(gb, 4);
  722|  28.6k|        hdr->quant.qm_v = seqhdr->separate_uv_delta_q ? dav1d_get_bits(gb, 4) :
  ------------------
  |  Branch (722:27): [True: 12.1k, False: 16.4k]
  ------------------
  723|  28.6k|                                                        hdr->quant.qm_u;
  724|  28.6k|    }
  725|       |#if DEBUG_FRAME_HDR
  726|       |    printf("HDR: post-qm: off=%td\n",
  727|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  728|       |#endif
  729|       |
  730|       |    // segmentation data
  731|   381k|    hdr->segmentation.enabled = dav1d_get_bit(gb);
  732|   381k|    if (hdr->segmentation.enabled) {
  ------------------
  |  Branch (732:9): [True: 23.0k, False: 358k]
  ------------------
  733|  23.0k|        if (hdr->primary_ref_frame == DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|  23.0k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (733:13): [True: 5.56k, False: 17.4k]
  ------------------
  734|  5.56k|            hdr->segmentation.update_map = 1;
  735|  5.56k|            hdr->segmentation.update_data = 1;
  736|  17.4k|        } else {
  737|  17.4k|            hdr->segmentation.update_map = dav1d_get_bit(gb);
  738|  17.4k|            if (hdr->segmentation.update_map)
  ------------------
  |  Branch (738:17): [True: 10.1k, False: 7.36k]
  ------------------
  739|  10.1k|                hdr->segmentation.temporal = dav1d_get_bit(gb);
  740|  17.4k|            hdr->segmentation.update_data = dav1d_get_bit(gb);
  741|  17.4k|        }
  742|       |
  743|  23.0k|        if (hdr->segmentation.update_data) {
  ------------------
  |  Branch (743:13): [True: 7.09k, False: 15.9k]
  ------------------
  744|  7.09k|            hdr->segmentation.seg_data.last_active_segid = -1;
  745|  63.8k|            for (int i = 0; i < DAV1D_MAX_SEGMENTS; i++) {
  ------------------
  |  |   43|  63.8k|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (745:29): [True: 56.7k, False: 7.09k]
  ------------------
  746|  56.7k|                Dav1dSegmentationData *const seg =
  747|  56.7k|                    &hdr->segmentation.seg_data.d[i];
  748|  56.7k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (748:21): [True: 13.2k, False: 43.5k]
  ------------------
  749|  13.2k|                    seg->delta_q = dav1d_get_sbits(gb, 9);
  750|  13.2k|                    hdr->segmentation.seg_data.last_active_segid = i;
  751|  13.2k|                }
  752|  56.7k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (752:21): [True: 10.5k, False: 46.2k]
  ------------------
  753|  10.5k|                    seg->delta_lf_y_v = dav1d_get_sbits(gb, 7);
  754|  10.5k|                    hdr->segmentation.seg_data.last_active_segid = i;
  755|  10.5k|                }
  756|  56.7k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (756:21): [True: 13.4k, False: 43.3k]
  ------------------
  757|  13.4k|                    seg->delta_lf_y_h = dav1d_get_sbits(gb, 7);
  758|  13.4k|                    hdr->segmentation.seg_data.last_active_segid = i;
  759|  13.4k|                }
  760|  56.7k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (760:21): [True: 11.9k, False: 44.8k]
  ------------------
  761|  11.9k|                    seg->delta_lf_u = dav1d_get_sbits(gb, 7);
  762|  11.9k|                    hdr->segmentation.seg_data.last_active_segid = i;
  763|  11.9k|                }
  764|  56.7k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (764:21): [True: 10.6k, False: 46.0k]
  ------------------
  765|  10.6k|                    seg->delta_lf_v = dav1d_get_sbits(gb, 7);
  766|  10.6k|                    hdr->segmentation.seg_data.last_active_segid = i;
  767|  10.6k|                }
  768|  56.7k|                if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (768:21): [True: 10.8k, False: 45.8k]
  ------------------
  769|  10.8k|                    seg->ref = dav1d_get_bits(gb, 3);
  770|  10.8k|                    hdr->segmentation.seg_data.last_active_segid = i;
  771|  10.8k|                    hdr->segmentation.seg_data.preskip = 1;
  772|  45.8k|                } else {
  773|  45.8k|                    seg->ref = -1;
  774|  45.8k|                }
  775|  56.7k|                if ((seg->skip = dav1d_get_bit(gb))) {
  ------------------
  |  Branch (775:21): [True: 12.2k, False: 44.4k]
  ------------------
  776|  12.2k|                    hdr->segmentation.seg_data.last_active_segid = i;
  777|  12.2k|                    hdr->segmentation.seg_data.preskip = 1;
  778|  12.2k|                }
  779|  56.7k|                if ((seg->globalmv = dav1d_get_bit(gb))) {
  ------------------
  |  Branch (779:21): [True: 11.9k, False: 44.8k]
  ------------------
  780|  11.9k|                    hdr->segmentation.seg_data.last_active_segid = i;
  781|  11.9k|                    hdr->segmentation.seg_data.preskip = 1;
  782|  11.9k|                }
  783|  56.7k|            }
  784|  15.9k|        } else {
  785|       |            // segmentation.update_data was false so we should copy
  786|       |            // segmentation data from the reference frame.
  787|  15.9k|            assert(hdr->primary_ref_frame != DAV1D_PRIMARY_REF_NONE);
  ------------------
  |  Branch (787:13): [True: 15.9k, False: 0]
  ------------------
  788|  15.9k|            const int pri_ref = hdr->refidx[hdr->primary_ref_frame];
  789|  15.9k|            if (!c->refs[pri_ref].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (789:17): [True: 239, False: 15.6k]
  ------------------
  790|  15.6k|            hdr->segmentation.seg_data =
  791|  15.6k|                c->refs[pri_ref].p.p.frame_hdr->segmentation.seg_data;
  792|  15.6k|        }
  793|   358k|    } else {
  794|  3.22M|        for (int i = 0; i < DAV1D_MAX_SEGMENTS; i++)
  ------------------
  |  |   43|  3.22M|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (794:25): [True: 2.86M, False: 358k]
  ------------------
  795|  2.86M|            hdr->segmentation.seg_data.d[i].ref = -1;
  796|   358k|    }
  797|       |#if DEBUG_FRAME_HDR
  798|       |    printf("HDR: post-segmentation: off=%td\n",
  799|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  800|       |#endif
  801|       |
  802|       |    // delta q
  803|   381k|    if (hdr->quant.yac) {
  ------------------
  |  Branch (803:9): [True: 339k, False: 41.8k]
  ------------------
  804|   339k|        hdr->delta.q.present = dav1d_get_bit(gb);
  805|   339k|        if (hdr->delta.q.present) {
  ------------------
  |  Branch (805:13): [True: 81.0k, False: 258k]
  ------------------
  806|  81.0k|            hdr->delta.q.res_log2 = dav1d_get_bits(gb, 2);
  807|  81.0k|            if (!hdr->allow_intrabc) {
  ------------------
  |  Branch (807:17): [True: 16.5k, False: 64.4k]
  ------------------
  808|  16.5k|                hdr->delta.lf.present = dav1d_get_bit(gb);
  809|  16.5k|                if (hdr->delta.lf.present) {
  ------------------
  |  Branch (809:21): [True: 8.21k, False: 8.37k]
  ------------------
  810|  8.21k|                    hdr->delta.lf.res_log2 = dav1d_get_bits(gb, 2);
  811|  8.21k|                    hdr->delta.lf.multi = dav1d_get_bit(gb);
  812|  8.21k|                }
  813|  16.5k|            }
  814|  81.0k|        }
  815|   339k|    }
  816|       |#if DEBUG_FRAME_HDR
  817|       |    printf("HDR: post-delta_q_lf_flags: off=%td\n",
  818|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  819|       |#endif
  820|       |
  821|       |    // derive lossless flags
  822|   381k|    const int delta_lossless = !hdr->quant.ydc_delta && !hdr->quant.udc_delta &&
  ------------------
  |  Branch (822:32): [True: 351k, False: 29.7k]
  |  Branch (822:57): [True: 344k, False: 6.30k]
  ------------------
  823|   344k|        !hdr->quant.uac_delta && !hdr->quant.vdc_delta && !hdr->quant.vac_delta;
  ------------------
  |  Branch (823:9): [True: 338k, False: 6.10k]
  |  Branch (823:34): [True: 338k, False: 552]
  |  Branch (823:59): [True: 338k, False: 17]
  ------------------
  824|   381k|    hdr->all_lossless = 1;
  825|  3.42M|    for (int i = 0; i < DAV1D_MAX_SEGMENTS; i++) {
  ------------------
  |  |   43|  3.42M|#define DAV1D_MAX_SEGMENTS 8
  ------------------
  |  Branch (825:21): [True: 3.04M, False: 381k]
  ------------------
  826|  3.04M|        hdr->segmentation.qidx[i] = hdr->segmentation.enabled ?
  ------------------
  |  Branch (826:37): [True: 182k, False: 2.86M]
  ------------------
  827|   182k|            iclip_u8(hdr->quant.yac + hdr->segmentation.seg_data.d[i].delta_q) :
  828|  3.04M|            hdr->quant.yac;
  829|  3.04M|        hdr->segmentation.lossless[i] =
  830|  3.04M|            !hdr->segmentation.qidx[i] && delta_lossless;
  ------------------
  |  Branch (830:13): [True: 340k, False: 2.70M]
  |  Branch (830:43): [True: 305k, False: 34.9k]
  ------------------
  831|  3.04M|        hdr->all_lossless &= hdr->segmentation.lossless[i];
  832|  3.04M|    }
  833|       |
  834|       |    // loopfilter
  835|   381k|    if (hdr->all_lossless || hdr->allow_intrabc) {
  ------------------
  |  Branch (835:9): [True: 37.8k, False: 343k]
  |  Branch (835:30): [True: 231k, False: 111k]
  ------------------
  836|   269k|        hdr->loopfilter.mode_ref_delta_enabled = 1;
  837|   269k|        hdr->loopfilter.mode_ref_delta_update = 1;
  838|   269k|        hdr->loopfilter.mode_ref_deltas = default_mode_ref_deltas;
  839|   269k|    } else {
  840|   111k|        hdr->loopfilter.level_y[0] = dav1d_get_bits(gb, 6);
  841|   111k|        hdr->loopfilter.level_y[1] = dav1d_get_bits(gb, 6);
  842|   111k|        if (!seqhdr->monochrome &&
  ------------------
  |  Branch (842:13): [True: 44.4k, False: 67.0k]
  ------------------
  843|  44.4k|            (hdr->loopfilter.level_y[0] || hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (843:14): [True: 19.4k, False: 25.0k]
  |  Branch (843:44): [True: 9.63k, False: 15.4k]
  ------------------
  844|  29.0k|        {
  845|  29.0k|            hdr->loopfilter.level_u = dav1d_get_bits(gb, 6);
  846|  29.0k|            hdr->loopfilter.level_v = dav1d_get_bits(gb, 6);
  847|  29.0k|        }
  848|   111k|        hdr->loopfilter.sharpness = dav1d_get_bits(gb, 3);
  849|       |
  850|   111k|        if (hdr->primary_ref_frame == DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|   111k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (850:13): [True: 49.3k, False: 62.1k]
  ------------------
  851|  49.3k|            hdr->loopfilter.mode_ref_deltas = default_mode_ref_deltas;
  852|  62.1k|        } else {
  853|  62.1k|            const int ref = hdr->refidx[hdr->primary_ref_frame];
  854|  62.1k|            if (!c->refs[ref].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (854:17): [True: 438, False: 61.7k]
  ------------------
  855|  61.7k|            hdr->loopfilter.mode_ref_deltas =
  856|  61.7k|                c->refs[ref].p.p.frame_hdr->loopfilter.mode_ref_deltas;
  857|  61.7k|        }
  858|   111k|        hdr->loopfilter.mode_ref_delta_enabled = dav1d_get_bit(gb);
  859|   111k|        if (hdr->loopfilter.mode_ref_delta_enabled) {
  ------------------
  |  Branch (859:13): [True: 65.1k, False: 45.9k]
  ------------------
  860|  65.1k|            hdr->loopfilter.mode_ref_delta_update = dav1d_get_bit(gb);
  861|  65.1k|            if (hdr->loopfilter.mode_ref_delta_update) {
  ------------------
  |  Branch (861:17): [True: 8.56k, False: 56.5k]
  ------------------
  862|  77.0k|                for (int i = 0; i < 8; i++)
  ------------------
  |  Branch (862:33): [True: 68.4k, False: 8.56k]
  ------------------
  863|  68.4k|                    if (dav1d_get_bit(gb))
  ------------------
  |  Branch (863:25): [True: 24.1k, False: 44.3k]
  ------------------
  864|  24.1k|                        hdr->loopfilter.mode_ref_deltas.ref_delta[i] =
  865|  24.1k|                            dav1d_get_sbits(gb, 7);
  866|  25.6k|                for (int i = 0; i < 2; i++)
  ------------------
  |  Branch (866:33): [True: 17.1k, False: 8.56k]
  ------------------
  867|  17.1k|                    if (dav1d_get_bit(gb))
  ------------------
  |  Branch (867:25): [True: 3.79k, False: 13.3k]
  ------------------
  868|  3.79k|                        hdr->loopfilter.mode_ref_deltas.mode_delta[i] =
  869|  3.79k|                            dav1d_get_sbits(gb, 7);
  870|  8.56k|            }
  871|  65.1k|        }
  872|   111k|    }
  873|       |#if DEBUG_FRAME_HDR
  874|       |    printf("HDR: post-lpf: off=%td\n",
  875|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  876|       |#endif
  877|       |
  878|       |    // cdef
  879|   380k|    if (!hdr->all_lossless && seqhdr->cdef && !hdr->allow_intrabc) {
  ------------------
  |  Branch (879:9): [True: 342k, False: 37.8k]
  |  Branch (879:31): [True: 229k, False: 113k]
  |  Branch (879:47): [True: 64.5k, False: 165k]
  ------------------
  880|  64.5k|        hdr->cdef.damping = dav1d_get_bits(gb, 2) + 3;
  881|  64.5k|        hdr->cdef.n_bits = dav1d_get_bits(gb, 2);
  882|   168k|        for (int i = 0; i < (1 << hdr->cdef.n_bits); i++) {
  ------------------
  |  Branch (882:25): [True: 103k, False: 64.5k]
  ------------------
  883|   103k|            hdr->cdef.y_strength[i] = dav1d_get_bits(gb, 6);
  884|   103k|            if (!seqhdr->monochrome)
  ------------------
  |  Branch (884:17): [True: 46.8k, False: 57.0k]
  ------------------
  885|  46.8k|                hdr->cdef.uv_strength[i] = dav1d_get_bits(gb, 6);
  886|   103k|        }
  887|  64.5k|    }
  888|       |#if DEBUG_FRAME_HDR
  889|       |    printf("HDR: post-cdef: off=%td\n",
  890|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  891|       |#endif
  892|       |
  893|       |    // restoration
  894|   380k|    if ((!hdr->all_lossless || hdr->super_res.enabled) &&
  ------------------
  |  Branch (894:10): [True: 342k, False: 37.8k]
  |  Branch (894:32): [True: 25.0k, False: 12.7k]
  ------------------
  895|   367k|        seqhdr->restoration && !hdr->allow_intrabc)
  ------------------
  |  Branch (895:9): [True: 105k, False: 262k]
  |  Branch (895:32): [True: 90.9k, False: 14.4k]
  ------------------
  896|  90.9k|    {
  897|  90.9k|        hdr->restoration.type[0] = dav1d_get_bits(gb, 2);
  898|  90.9k|        if (!seqhdr->monochrome) {
  ------------------
  |  Branch (898:13): [True: 37.7k, False: 53.2k]
  ------------------
  899|  37.7k|            hdr->restoration.type[1] = dav1d_get_bits(gb, 2);
  900|  37.7k|            hdr->restoration.type[2] = dav1d_get_bits(gb, 2);
  901|  37.7k|        }
  902|       |
  903|  90.9k|        if (hdr->restoration.type[0] || hdr->restoration.type[1] ||
  ------------------
  |  Branch (903:13): [True: 55.1k, False: 35.8k]
  |  Branch (903:41): [True: 3.81k, False: 32.0k]
  ------------------
  904|  32.0k|            hdr->restoration.type[2])
  ------------------
  |  Branch (904:13): [True: 944, False: 31.1k]
  ------------------
  905|  59.8k|        {
  906|       |            // Log2 of the restoration unit size.
  907|  59.8k|            hdr->restoration.unit_size[0] = 6 + seqhdr->sb128;
  908|  59.8k|            if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (908:17): [True: 11.2k, False: 48.6k]
  ------------------
  909|  11.2k|                hdr->restoration.unit_size[0]++;
  910|  11.2k|                if (!seqhdr->sb128)
  ------------------
  |  Branch (910:21): [True: 3.01k, False: 8.19k]
  ------------------
  911|  3.01k|                    hdr->restoration.unit_size[0] += dav1d_get_bit(gb);
  912|  11.2k|            }
  913|  59.8k|            hdr->restoration.unit_size[1] = hdr->restoration.unit_size[0];
  914|  59.8k|            if ((hdr->restoration.type[1] || hdr->restoration.type[2]) &&
  ------------------
  |  Branch (914:18): [True: 10.9k, False: 48.9k]
  |  Branch (914:46): [True: 4.46k, False: 44.4k]
  ------------------
  915|  15.4k|                seqhdr->ss_hor == 1 && seqhdr->ss_ver == 1)
  ------------------
  |  Branch (915:17): [True: 10.3k, False: 5.07k]
  |  Branch (915:40): [True: 9.55k, False: 774]
  ------------------
  916|  9.55k|            {
  917|  9.55k|                hdr->restoration.unit_size[1] -= dav1d_get_bit(gb);
  918|  9.55k|            }
  919|  59.8k|        } else {
  920|  31.1k|            hdr->restoration.unit_size[0] = 8;
  921|  31.1k|        }
  922|  90.9k|    }
  923|       |#if DEBUG_FRAME_HDR
  924|       |    printf("HDR: post-restoration: off=%td\n",
  925|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  926|       |#endif
  927|       |
  928|   380k|    if (!hdr->all_lossless)
  ------------------
  |  Branch (928:9): [True: 342k, False: 37.8k]
  ------------------
  929|   342k|        hdr->txfm_mode = dav1d_get_bit(gb) ? DAV1D_TX_SWITCHABLE : DAV1D_TX_LARGEST;
  ------------------
  |  Branch (929:26): [True: 192k, False: 149k]
  ------------------
  930|       |#if DEBUG_FRAME_HDR
  931|       |    printf("HDR: post-txfmmode: off=%td\n",
  932|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  933|       |#endif
  934|   380k|    if (IS_INTER_OR_SWITCH(hdr))
  ------------------
  |  |   36|   380k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 110k, False: 270k]
  |  |  ------------------
  ------------------
  935|   110k|        hdr->switchable_comp_refs = dav1d_get_bit(gb);
  936|       |#if DEBUG_FRAME_HDR
  937|       |    printf("HDR: post-refmode: off=%td\n",
  938|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  939|       |#endif
  940|   380k|    if (hdr->switchable_comp_refs && IS_INTER_OR_SWITCH(hdr) && seqhdr->order_hint) {
  ------------------
  |  |   36|   437k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 56.5k, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (940:9): [True: 56.5k, False: 324k]
  |  Branch (940:65): [True: 54.8k, False: 1.75k]
  ------------------
  941|  54.8k|        const int poc = hdr->frame_offset;
  942|  54.8k|        int off_before = -1, off_after = -1;
  943|  54.8k|        int off_before_idx, off_after_idx;
  944|   437k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (944:25): [True: 383k, False: 54.7k]
  ------------------
  945|   383k|            if (!c->refs[hdr->refidx[i]].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (945:17): [True: 85, False: 383k]
  ------------------
  946|   383k|            const int refpoc = c->refs[hdr->refidx[i]].p.p.frame_hdr->frame_offset;
  947|       |
  948|   383k|            const int diff = get_poc_diff(seqhdr->order_hint_n_bits, refpoc, poc);
  949|   383k|            if (diff > 0) {
  ------------------
  |  Branch (949:17): [True: 156k, False: 226k]
  ------------------
  950|   156k|                if (off_after < 0 || get_poc_diff(seqhdr->order_hint_n_bits,
  ------------------
  |  Branch (950:21): [True: 45.1k, False: 111k]
  |  Branch (950:38): [True: 9.00k, False: 102k]
  ------------------
  951|   111k|                                                  off_after, refpoc) > 0)
  952|  54.1k|                {
  953|  54.1k|                    off_after = refpoc;
  954|  54.1k|                    off_after_idx = i;
  955|  54.1k|                }
  956|   226k|            } else if (diff < 0 && (off_before < 0 ||
  ------------------
  |  Branch (956:24): [True: 169k, False: 57.0k]
  |  Branch (956:37): [True: 49.6k, False: 119k]
  ------------------
  957|   119k|                                    get_poc_diff(seqhdr->order_hint_n_bits,
  ------------------
  |  Branch (957:37): [True: 6.87k, False: 112k]
  ------------------
  958|   119k|                                                 refpoc, off_before) > 0))
  959|  56.5k|            {
  960|  56.5k|                off_before = refpoc;
  961|  56.5k|                off_before_idx = i;
  962|  56.5k|            }
  963|   383k|        }
  964|       |
  965|  54.7k|        if ((off_before | off_after) >= 0) {
  ------------------
  |  Branch (965:13): [True: 41.2k, False: 13.5k]
  ------------------
  966|  41.2k|            hdr->skip_mode_refs[0] = imin(off_before_idx, off_after_idx);
  967|  41.2k|            hdr->skip_mode_refs[1] = imax(off_before_idx, off_after_idx);
  968|  41.2k|            hdr->skip_mode_allowed = 1;
  969|  41.2k|        } else if (off_before >= 0) {
  ------------------
  |  Branch (969:20): [True: 8.46k, False: 5.05k]
  ------------------
  970|  8.46k|            int off_before2 = -1;
  971|  8.46k|            int off_before2_idx;
  972|  67.6k|            for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (972:29): [True: 59.2k, False: 8.46k]
  ------------------
  973|  59.2k|                if (!c->refs[hdr->refidx[i]].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (973:21): [True: 0, False: 59.2k]
  ------------------
  974|  59.2k|                const int refpoc = c->refs[hdr->refidx[i]].p.p.frame_hdr->frame_offset;
  975|  59.2k|                if (get_poc_diff(seqhdr->order_hint_n_bits,
  ------------------
  |  Branch (975:21): [True: 16.0k, False: 43.1k]
  ------------------
  976|  59.2k|                                 refpoc, off_before) < 0) {
  977|  16.0k|                    if (off_before2 < 0 || get_poc_diff(seqhdr->order_hint_n_bits,
  ------------------
  |  Branch (977:25): [True: 4.56k, False: 11.4k]
  |  Branch (977:44): [True: 483, False: 10.9k]
  ------------------
  978|  11.4k|                                                        refpoc, off_before2) > 0)
  979|  5.04k|                    {
  980|  5.04k|                        off_before2 = refpoc;
  981|  5.04k|                        off_before2_idx = i;
  982|  5.04k|                    }
  983|  16.0k|                }
  984|  59.2k|            }
  985|       |
  986|  8.46k|            if (off_before2 >= 0) {
  ------------------
  |  Branch (986:17): [True: 4.56k, False: 3.89k]
  ------------------
  987|  4.56k|                hdr->skip_mode_refs[0] = imin(off_before_idx, off_before2_idx);
  988|  4.56k|                hdr->skip_mode_refs[1] = imax(off_before_idx, off_before2_idx);
  989|  4.56k|                hdr->skip_mode_allowed = 1;
  990|  4.56k|            }
  991|  8.46k|        }
  992|  54.7k|    }
  993|   380k|    if (hdr->skip_mode_allowed)
  ------------------
  |  Branch (993:9): [True: 45.7k, False: 334k]
  ------------------
  994|  45.7k|        hdr->skip_mode_enabled = dav1d_get_bit(gb);
  995|       |#if DEBUG_FRAME_HDR
  996|       |    printf("HDR: post-extskip: off=%td\n",
  997|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
  998|       |#endif
  999|   380k|    if (!hdr->error_resilient_mode && IS_INTER_OR_SWITCH(hdr) && seqhdr->warped_motion)
  ------------------
  |  |   36|   491k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 108k, False: 2.42k]
  |  |  ------------------
  ------------------
  |  Branch (999:9): [True: 110k, False: 269k]
  |  Branch (999:66): [True: 89.9k, False: 18.2k]
  ------------------
 1000|  89.9k|        hdr->warp_motion = dav1d_get_bit(gb);
 1001|       |#if DEBUG_FRAME_HDR
 1002|       |    printf("HDR: post-warpmotionbit: off=%td\n",
 1003|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
 1004|       |#endif
 1005|   380k|    hdr->reduced_txtp_set = dav1d_get_bit(gb);
 1006|       |#if DEBUG_FRAME_HDR
 1007|       |    printf("HDR: post-reducedtxtpset: off=%td\n",
 1008|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
 1009|       |#endif
 1010|       |
 1011|  3.04M|    for (int i = 0; i < 7; i++)
  ------------------
  |  Branch (1011:21): [True: 2.66M, False: 380k]
  ------------------
 1012|  2.66M|        hdr->gmv[i] = dav1d_default_wm_params;
 1013|       |
 1014|   380k|    if (IS_INTER_OR_SWITCH(hdr)) {
  ------------------
  |  |   36|   380k|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (36:5): [True: 110k, False: 270k]
  |  |  ------------------
  ------------------
 1015|   881k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (1015:25): [True: 771k, False: 109k]
  ------------------
 1016|   771k|            hdr->gmv[i].type = !dav1d_get_bit(gb) ? DAV1D_WM_TYPE_IDENTITY :
  ------------------
  |  Branch (1016:32): [True: 731k, False: 39.8k]
  ------------------
 1017|   771k|                                dav1d_get_bit(gb) ? DAV1D_WM_TYPE_ROT_ZOOM :
  ------------------
  |  Branch (1017:33): [True: 22.4k, False: 17.4k]
  ------------------
 1018|  39.8k|                                dav1d_get_bit(gb) ? DAV1D_WM_TYPE_TRANSLATION :
  ------------------
  |  Branch (1018:33): [True: 3.45k, False: 13.9k]
  ------------------
 1019|  17.4k|                                                    DAV1D_WM_TYPE_AFFINE;
 1020|       |
 1021|   771k|            if (hdr->gmv[i].type == DAV1D_WM_TYPE_IDENTITY) continue;
  ------------------
  |  Branch (1021:17): [True: 731k, False: 39.8k]
  ------------------
 1022|       |
 1023|  39.8k|            const Dav1dWarpedMotionParams *ref_gmv;
 1024|  39.8k|            if (hdr->primary_ref_frame == DAV1D_PRIMARY_REF_NONE) {
  ------------------
  |  |   45|  39.8k|#define DAV1D_PRIMARY_REF_NONE 7
  ------------------
  |  Branch (1024:17): [True: 2.69k, False: 37.1k]
  ------------------
 1025|  2.69k|                ref_gmv = &dav1d_default_wm_params;
 1026|  37.1k|            } else {
 1027|  37.1k|                const int pri_ref = hdr->refidx[hdr->primary_ref_frame];
 1028|  37.1k|                if (!c->refs[pri_ref].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (1028:21): [True: 488, False: 36.6k]
  ------------------
 1029|  36.6k|                ref_gmv = &c->refs[pri_ref].p.p.frame_hdr->gmv[i];
 1030|  36.6k|            }
 1031|  39.3k|            int32_t *const mat = hdr->gmv[i].matrix;
 1032|  39.3k|            const int32_t *const ref_mat = ref_gmv->matrix;
 1033|  39.3k|            int bits, shift;
 1034|       |
 1035|  39.3k|            if (hdr->gmv[i].type >= DAV1D_WM_TYPE_ROT_ZOOM) {
  ------------------
  |  Branch (1035:17): [True: 36.0k, False: 3.35k]
  ------------------
 1036|  36.0k|                mat[2] = (1 << 16) + 2 *
 1037|  36.0k|                    dav1d_get_bits_subexp(gb, (ref_mat[2] - (1 << 16)) >> 1, 12);
 1038|  36.0k|                mat[3] = 2 * dav1d_get_bits_subexp(gb, ref_mat[3] >> 1, 12);
 1039|       |
 1040|  36.0k|                bits = 12;
 1041|  36.0k|                shift = 10;
 1042|  36.0k|            } else {
 1043|  3.35k|                bits = 9 - !hdr->hp;
 1044|  3.35k|                shift = 13 + !hdr->hp;
 1045|  3.35k|            }
 1046|       |
 1047|  39.3k|            if (hdr->gmv[i].type == DAV1D_WM_TYPE_AFFINE) {
  ------------------
  |  Branch (1047:17): [True: 13.9k, False: 25.4k]
  ------------------
 1048|  13.9k|                mat[4] = 2 * dav1d_get_bits_subexp(gb, ref_mat[4] >> 1, 12);
 1049|  13.9k|                mat[5] = (1 << 16) + 2 *
 1050|  13.9k|                    dav1d_get_bits_subexp(gb, (ref_mat[5] - (1 << 16)) >> 1, 12);
 1051|  25.4k|            } else {
 1052|  25.4k|                mat[4] = -mat[3];
 1053|  25.4k|                mat[5] = mat[2];
 1054|  25.4k|            }
 1055|       |
 1056|  39.3k|            mat[0] = dav1d_get_bits_subexp(gb, ref_mat[0] >> shift, bits) * (1 << shift);
 1057|  39.3k|            mat[1] = dav1d_get_bits_subexp(gb, ref_mat[1] >> shift, bits) * (1 << shift);
 1058|  39.3k|        }
 1059|   110k|    }
 1060|       |#if DEBUG_FRAME_HDR
 1061|       |    printf("HDR: post-gmv: off=%td\n",
 1062|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
 1063|       |#endif
 1064|       |
 1065|   380k|    if (seqhdr->film_grain_present && (hdr->show_frame || hdr->showable_frame)) {
  ------------------
  |  Branch (1065:9): [True: 292k, False: 87.5k]
  |  Branch (1065:40): [True: 247k, False: 45.0k]
  |  Branch (1065:59): [True: 1.36k, False: 43.6k]
  ------------------
 1066|   248k|        hdr->film_grain.present = dav1d_get_bit(gb);
 1067|   248k|        if (hdr->film_grain.present) {
  ------------------
  |  Branch (1067:13): [True: 25.9k, False: 222k]
  ------------------
 1068|  25.9k|            const unsigned seed = dav1d_get_bits(gb, 16);
 1069|  25.9k|            hdr->film_grain.update = hdr->frame_type != DAV1D_FRAME_TYPE_INTER || dav1d_get_bit(gb);
  ------------------
  |  Branch (1069:38): [True: 6.95k, False: 19.0k]
  |  Branch (1069:83): [True: 747, False: 18.2k]
  ------------------
 1070|  25.9k|            if (!hdr->film_grain.update) {
  ------------------
  |  Branch (1070:17): [True: 18.2k, False: 7.70k]
  ------------------
 1071|  18.2k|                const int refidx = dav1d_get_bits(gb, 3);
 1072|  18.2k|                int i;
 1073|  45.0k|                for (i = 0; i < 7; i++)
  ------------------
  |  Branch (1073:29): [True: 44.8k, False: 254]
  ------------------
 1074|  44.8k|                    if (hdr->refidx[i] == refidx)
  ------------------
  |  Branch (1074:25): [True: 18.0k, False: 26.8k]
  ------------------
 1075|  18.0k|                        break;
 1076|  18.2k|                if (i == 7 || !c->refs[refidx].p.p.frame_hdr) goto error;
  ------------------
  |  Branch (1076:21): [True: 254, False: 18.0k]
  |  Branch (1076:31): [True: 73, False: 17.9k]
  ------------------
 1077|  17.9k|                hdr->film_grain.data = c->refs[refidx].p.p.frame_hdr->film_grain.data;
 1078|  17.9k|                hdr->film_grain.data.seed = seed;
 1079|  17.9k|            } else {
 1080|  7.70k|                Dav1dFilmGrainData *const fgd = &hdr->film_grain.data;
 1081|  7.70k|                fgd->seed = seed;
 1082|       |
 1083|  7.70k|                fgd->num_y_points = dav1d_get_bits(gb, 4);
 1084|  7.70k|                if (fgd->num_y_points > 14) goto error;
  ------------------
  |  Branch (1084:21): [True: 301, False: 7.40k]
  ------------------
 1085|  11.5k|                for (int i = 0; i < fgd->num_y_points; i++) {
  ------------------
  |  Branch (1085:33): [True: 4.98k, False: 6.61k]
  ------------------
 1086|  4.98k|                    fgd->y_points[i][0] = dav1d_get_bits(gb, 8);
 1087|  4.98k|                    if (i && fgd->y_points[i - 1][0] >= fgd->y_points[i][0])
  ------------------
  |  Branch (1087:25): [True: 2.64k, False: 2.34k]
  |  Branch (1087:30): [True: 791, False: 1.84k]
  ------------------
 1088|    791|                        goto error;
 1089|  4.19k|                    fgd->y_points[i][1] = dav1d_get_bits(gb, 8);
 1090|  4.19k|                }
 1091|       |
 1092|  6.61k|                if (!seqhdr->monochrome)
  ------------------
  |  Branch (1092:21): [True: 5.76k, False: 847]
  ------------------
 1093|  5.76k|                    fgd->chroma_scaling_from_luma = dav1d_get_bit(gb);
 1094|  6.61k|                if (seqhdr->monochrome || fgd->chroma_scaling_from_luma ||
  ------------------
  |  Branch (1094:21): [True: 847, False: 5.76k]
  |  Branch (1094:43): [True: 2.38k, False: 3.37k]
  ------------------
 1095|  3.37k|                    (seqhdr->ss_ver == 1 && seqhdr->ss_hor == 1 && !fgd->num_y_points))
  ------------------
  |  Branch (1095:22): [True: 755, False: 2.62k]
  |  Branch (1095:45): [True: 755, False: 0]
  |  Branch (1095:68): [True: 213, False: 542]
  ------------------
 1096|  3.44k|                {
 1097|  3.44k|                    fgd->num_uv_points[0] = fgd->num_uv_points[1] = 0;
 1098|  8.64k|                } else for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1098:41): [True: 5.97k, False: 2.66k]
  ------------------
 1099|  5.97k|                    fgd->num_uv_points[pl] = dav1d_get_bits(gb, 4);
 1100|  5.97k|                    if (fgd->num_uv_points[pl] > 10) goto error;
  ------------------
  |  Branch (1100:25): [True: 85, False: 5.89k]
  ------------------
 1101|  10.8k|                    for (int i = 0; i < fgd->num_uv_points[pl]; i++) {
  ------------------
  |  Branch (1101:37): [True: 5.41k, False: 5.47k]
  ------------------
 1102|  5.41k|                        fgd->uv_points[pl][i][0] = dav1d_get_bits(gb, 8);
 1103|  5.41k|                        if (i && fgd->uv_points[pl][i - 1][0] >= fgd->uv_points[pl][i][0])
  ------------------
  |  Branch (1103:29): [True: 3.35k, False: 2.06k]
  |  Branch (1103:34): [True: 418, False: 2.93k]
  ------------------
 1104|    418|                            goto error;
 1105|  4.99k|                        fgd->uv_points[pl][i][1] = dav1d_get_bits(gb, 8);
 1106|  4.99k|                    }
 1107|  5.89k|                }
 1108|       |
 1109|  6.10k|                if (seqhdr->ss_hor == 1 && seqhdr->ss_ver == 1 &&
  ------------------
  |  Branch (1109:21): [True: 4.33k, False: 1.76k]
  |  Branch (1109:44): [True: 3.62k, False: 717]
  ------------------
 1110|  3.62k|                    !!fgd->num_uv_points[0] != !!fgd->num_uv_points[1])
  ------------------
  |  Branch (1110:21): [True: 70, False: 3.55k]
  ------------------
 1111|     70|                {
 1112|     70|                    goto error;
 1113|     70|                }
 1114|       |
 1115|  6.03k|                fgd->scaling_shift = dav1d_get_bits(gb, 2) + 8;
 1116|  6.03k|                fgd->ar_coeff_lag = dav1d_get_bits(gb, 2);
 1117|  6.03k|                const int num_y_pos = 2 * fgd->ar_coeff_lag * (fgd->ar_coeff_lag + 1);
 1118|  6.03k|                if (fgd->num_y_points)
  ------------------
  |  Branch (1118:21): [True: 1.42k, False: 4.61k]
  ------------------
 1119|  21.0k|                    for (int i = 0; i < num_y_pos; i++)
  ------------------
  |  Branch (1119:37): [True: 19.6k, False: 1.42k]
  ------------------
 1120|  19.6k|                        fgd->ar_coeffs_y[i] = dav1d_get_bits(gb, 8) - 128;
 1121|  18.1k|                for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (1121:34): [True: 12.0k, False: 6.03k]
  ------------------
 1122|  12.0k|                    if (fgd->num_uv_points[pl] || fgd->chroma_scaling_from_luma) {
  ------------------
  |  Branch (1122:25): [True: 1.56k, False: 10.5k]
  |  Branch (1122:51): [True: 4.76k, False: 5.73k]
  ------------------
 1123|  6.33k|                        const int num_uv_pos = num_y_pos + !!fgd->num_y_points;
 1124|  28.4k|                        for (int i = 0; i < num_uv_pos; i++)
  ------------------
  |  Branch (1124:41): [True: 22.1k, False: 6.33k]
  ------------------
 1125|  22.1k|                            fgd->ar_coeffs_uv[pl][i] = dav1d_get_bits(gb, 8) - 128;
 1126|  6.33k|                        if (!fgd->num_y_points)
  ------------------
  |  Branch (1126:29): [True: 5.00k, False: 1.32k]
  ------------------
 1127|  5.00k|                            fgd->ar_coeffs_uv[pl][num_uv_pos] = 0;
 1128|  6.33k|                    }
 1129|  6.03k|                fgd->ar_coeff_shift = dav1d_get_bits(gb, 2) + 6;
 1130|  6.03k|                fgd->grain_scale_shift = dav1d_get_bits(gb, 2);
 1131|  18.1k|                for (int pl = 0; pl < 2; pl++)
  ------------------
  |  Branch (1131:34): [True: 12.0k, False: 6.03k]
  ------------------
 1132|  12.0k|                    if (fgd->num_uv_points[pl]) {
  ------------------
  |  Branch (1132:25): [True: 1.56k, False: 10.5k]
  ------------------
 1133|  1.56k|                        fgd->uv_mult[pl] = dav1d_get_bits(gb, 8) - 128;
 1134|  1.56k|                        fgd->uv_luma_mult[pl] = dav1d_get_bits(gb, 8) - 128;
 1135|  1.56k|                        fgd->uv_offset[pl] = dav1d_get_bits(gb, 9) - 256;
 1136|  1.56k|                    }
 1137|  6.03k|                fgd->overlap_flag = dav1d_get_bit(gb);
 1138|  6.03k|                fgd->clip_to_restricted_range = dav1d_get_bit(gb);
 1139|  6.03k|            }
 1140|  25.9k|        }
 1141|   248k|    }
 1142|       |#if DEBUG_FRAME_HDR
 1143|       |    printf("HDR: post-filmgrain: off=%td\n",
 1144|       |           (gb->ptr - init_ptr) * 8 - gb->bits_left);
 1145|       |#endif
 1146|       |
 1147|   378k|    return 0;
 1148|       |
 1149|  7.38k|error:
 1150|  7.38k|    dav1d_log(c, "Error parsing frame header\n");
  ------------------
  |  |   44|  7.38k|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 7.38k]
  |  |  ------------------
  ------------------
 1151|  7.38k|    return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  7.38k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
 1152|   380k|}
obu.c:read_frame_size:
  343|   381k|{
  344|   381k|    const Dav1dSequenceHeader *const seqhdr = c->seq_hdr;
  345|   381k|    Dav1dFrameHeader *const hdr = c->frame_hdr;
  346|       |
  347|   381k|    if (use_ref) {
  ------------------
  |  Branch (347:9): [True: 62.0k, False: 319k]
  ------------------
  348|   111k|        for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (348:25): [True: 110k, False: 807]
  ------------------
  349|   110k|            if (dav1d_get_bit(gb)) {
  ------------------
  |  Branch (349:17): [True: 61.2k, False: 49.6k]
  ------------------
  350|  61.2k|                const Dav1dThreadPicture *const ref =
  351|  61.2k|                    &c->refs[c->frame_hdr->refidx[i]].p;
  352|  61.2k|                if (!ref->p.frame_hdr) return -1;
  ------------------
  |  Branch (352:21): [True: 259, False: 61.0k]
  ------------------
  353|  61.0k|                hdr->width[1] = ref->p.frame_hdr->width[1];
  354|  61.0k|                hdr->height = ref->p.frame_hdr->height;
  355|  61.0k|                hdr->render_width = ref->p.frame_hdr->render_width;
  356|  61.0k|                hdr->render_height = ref->p.frame_hdr->render_height;
  357|  61.0k|                hdr->super_res.enabled = seqhdr->super_res && dav1d_get_bit(gb);
  ------------------
  |  Branch (357:42): [True: 6.34k, False: 54.6k]
  |  Branch (357:63): [True: 3.59k, False: 2.75k]
  ------------------
  358|  61.0k|                if (hdr->super_res.enabled) {
  ------------------
  |  Branch (358:21): [True: 3.59k, False: 57.4k]
  ------------------
  359|  3.59k|                    const int d = hdr->super_res.width_scale_denominator =
  360|  3.59k|                        9 + dav1d_get_bits(gb, 3);
  361|  3.59k|                    hdr->width[0] = imax((hdr->width[1] * 8 + (d >> 1)) / d,
  362|  3.59k|                                         imin(16, hdr->width[1]));
  363|  57.4k|                } else {
  364|  57.4k|                    hdr->super_res.width_scale_denominator = 8;
  365|  57.4k|                    hdr->width[0] = hdr->width[1];
  366|  57.4k|                }
  367|  61.0k|                return 0;
  368|  61.2k|            }
  369|   110k|        }
  370|  62.0k|    }
  371|       |
  372|   320k|    if (hdr->frame_size_override) {
  ------------------
  |  Branch (372:9): [True: 9.45k, False: 311k]
  ------------------
  373|  9.45k|        hdr->width[1] = dav1d_get_bits(gb, seqhdr->width_n_bits) + 1;
  374|  9.45k|        hdr->height = dav1d_get_bits(gb, seqhdr->height_n_bits) + 1;
  375|   311k|    } else {
  376|   311k|        hdr->width[1] = seqhdr->max_width;
  377|   311k|        hdr->height = seqhdr->max_height;
  378|   311k|    }
  379|   320k|    hdr->super_res.enabled = seqhdr->super_res && dav1d_get_bit(gb);
  ------------------
  |  Branch (379:30): [True: 287k, False: 33.6k]
  |  Branch (379:51): [True: 48.2k, False: 238k]
  ------------------
  380|   320k|    if (hdr->super_res.enabled) {
  ------------------
  |  Branch (380:9): [True: 48.2k, False: 272k]
  ------------------
  381|  48.2k|        const int d = hdr->super_res.width_scale_denominator = 9 + dav1d_get_bits(gb, 3);
  382|  48.2k|        hdr->width[0] = imax((hdr->width[1] * 8 + (d >> 1)) / d, imin(16, hdr->width[1]));
  383|   272k|    } else {
  384|   272k|        hdr->super_res.width_scale_denominator = 8;
  385|   272k|        hdr->width[0] = hdr->width[1];
  386|   272k|    }
  387|   320k|    hdr->have_render_size = dav1d_get_bit(gb);
  388|   320k|    if (hdr->have_render_size) {
  ------------------
  |  Branch (388:9): [True: 8.69k, False: 311k]
  ------------------
  389|  8.69k|        hdr->render_width = dav1d_get_bits(gb, 16) + 1;
  390|  8.69k|        hdr->render_height = dav1d_get_bits(gb, 16) + 1;
  391|   311k|    } else {
  392|   311k|        hdr->render_width = hdr->width[1];
  393|   311k|        hdr->render_height = hdr->height;
  394|   311k|    }
  395|   320k|    return 0;
  396|   381k|}
obu.c:tile_log2:
  398|  1.63M|static inline int tile_log2(const int sz, const int tgt) {
  399|  1.63M|    int k;
  400|  2.38M|    for (k = 0; (sz << k) < tgt; k++) ;
  ------------------
  |  Branch (400:17): [True: 745k, False: 1.63M]
  ------------------
  401|  1.63M|    return k;
  402|  1.63M|}
obu.c:check_trailing_bits:
   50|  75.4k|{
   51|  75.4k|    const int trailing_one_bit = dav1d_get_bit(gb);
   52|       |
   53|  75.4k|    if (gb->error)
  ------------------
  |  Branch (53:9): [True: 3.75k, False: 71.7k]
  ------------------
   54|  3.75k|        return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|  3.75k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
   55|       |
   56|  71.7k|    if (!strict_std_compliance)
  ------------------
  |  Branch (56:9): [True: 71.7k, False: 0]
  ------------------
   57|  71.7k|        return 0;
   58|       |
   59|      0|    if (!trailing_one_bit || gb->state)
  ------------------
  |  Branch (59:9): [True: 0, False: 0]
  |  Branch (59:30): [True: 0, False: 0]
  ------------------
   60|      0|        return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
   61|       |
   62|      0|    ptrdiff_t size = gb->ptr_end - gb->ptr;
   63|      0|    while (size > 0 && gb->ptr[size - 1] == 0)
  ------------------
  |  Branch (63:12): [True: 0, False: 0]
  |  Branch (63:24): [True: 0, False: 0]
  ------------------
   64|      0|        size--;
   65|       |
   66|      0|    if (size)
  ------------------
  |  Branch (66:9): [True: 0, False: 0]
  ------------------
   67|      0|        return DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
   68|       |
   69|      0|    return 0;
   70|      0|}
obu.c:parse_tile_hdr:
 1154|   374k|static void parse_tile_hdr(Dav1dContext *const c, GetBits *const gb) {
 1155|   374k|    const int n_tiles = c->frame_hdr->tiling.cols * c->frame_hdr->tiling.rows;
 1156|   374k|    const int have_tile_pos = n_tiles > 1 ? dav1d_get_bit(gb) : 0;
  ------------------
  |  Branch (1156:31): [True: 34.6k, False: 339k]
  ------------------
 1157|       |
 1158|   374k|    if (have_tile_pos) {
  ------------------
  |  Branch (1158:9): [True: 2.61k, False: 371k]
  ------------------
 1159|  2.61k|        const int n_bits = c->frame_hdr->tiling.log2_cols +
 1160|  2.61k|                           c->frame_hdr->tiling.log2_rows;
 1161|  2.61k|        c->tile[c->n_tile_data].start = dav1d_get_bits(gb, n_bits);
 1162|  2.61k|        c->tile[c->n_tile_data].end = dav1d_get_bits(gb, n_bits);
 1163|   371k|    } else {
 1164|   371k|        c->tile[c->n_tile_data].start = 0;
 1165|   371k|        c->tile[c->n_tile_data].end = n_tiles - 1;
 1166|   371k|    }
 1167|   374k|}

dav1d_pal_dsp_init:
   71|  9.41k|COLD void dav1d_pal_dsp_init(Dav1dPalDSPContext *const c) {
   72|  9.41k|    c->pal_idx_finish = pal_idx_finish_c;
   73|       |
   74|  9.41k|#if HAVE_ASM
   75|       |#if ARCH_RISCV
   76|       |    pal_dsp_init_riscv(c);
   77|       |#elif ARCH_X86
   78|       |    pal_dsp_init_x86(c);
   79|  9.41k|#endif
   80|  9.41k|#endif
   81|  9.41k|}

dav1d_default_picture_alloc:
   46|   405k|int dav1d_default_picture_alloc(Dav1dPicture *const p, void *const cookie) {
   47|   405k|    const int hbd = p->p.bpc > 8;
   48|   405k|    const int aligned_w = (p->p.w + 127) & ~127;
   49|   405k|    const int aligned_h = (p->p.h + 127) & ~127;
   50|   405k|    const int has_chroma = p->p.layout != DAV1D_PIXEL_LAYOUT_I400;
   51|   405k|    const int ss_ver = p->p.layout == DAV1D_PIXEL_LAYOUT_I420;
   52|   405k|    const int ss_hor = p->p.layout != DAV1D_PIXEL_LAYOUT_I444;
   53|   405k|    ptrdiff_t y_stride = aligned_w << hbd;
   54|   405k|    ptrdiff_t uv_stride = has_chroma ? y_stride >> ss_hor : 0;
  ------------------
  |  Branch (54:27): [True: 328k, False: 77.0k]
  ------------------
   55|       |    /* Due to how mapping of addresses to sets works in most L1 and L2 cache
   56|       |     * implementations, strides of multiples of certain power-of-two numbers
   57|       |     * may cause multiple rows of the same superblock to map to the same set,
   58|       |     * causing evictions of previous rows resulting in a reduction in cache
   59|       |     * hit rate. Avoid that by slightly padding the stride when necessary. */
   60|   405k|    if (!(y_stride & 1023))
  ------------------
  |  Branch (60:9): [True: 39.1k, False: 366k]
  ------------------
   61|  39.1k|        y_stride += DAV1D_PICTURE_ALIGNMENT;
  ------------------
  |  |   44|  39.1k|#define DAV1D_PICTURE_ALIGNMENT 64
  ------------------
   62|   405k|    if (!(uv_stride & 1023) && has_chroma)
  ------------------
  |  Branch (62:9): [True: 113k, False: 291k]
  |  Branch (62:32): [True: 36.6k, False: 77.0k]
  ------------------
   63|  36.6k|        uv_stride += DAV1D_PICTURE_ALIGNMENT;
  ------------------
  |  |   44|  36.6k|#define DAV1D_PICTURE_ALIGNMENT 64
  ------------------
   64|   405k|    p->stride[0] = y_stride;
   65|   405k|    p->stride[1] = uv_stride;
   66|   405k|    const size_t y_sz = y_stride * aligned_h;
   67|   405k|    const size_t uv_sz = uv_stride * (aligned_h >> ss_ver);
   68|   405k|    const size_t pic_size = y_sz + 2 * uv_sz;
   69|       |
   70|   405k|    uint8_t *const buf = dav1d_mem_pool_pop(cookie, pic_size + DAV1D_PICTURE_ALIGNMENT);
  ------------------
  |  |   44|   405k|#define DAV1D_PICTURE_ALIGNMENT 64
  ------------------
   71|   405k|    if (!buf) return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (71:9): [True: 0, False: 405k]
  ------------------
   72|   405k|    p->allocator_data = buf;
   73|   405k|    p->data[0] = buf;
   74|   405k|    p->data[1] = has_chroma ? buf + y_sz : NULL;
  ------------------
  |  Branch (74:18): [True: 328k, False: 77.0k]
  ------------------
   75|   405k|    p->data[2] = has_chroma ? buf + y_sz + uv_sz : NULL;
  ------------------
  |  Branch (75:18): [True: 328k, False: 77.0k]
  ------------------
   76|       |
   77|   405k|    return 0;
   78|   405k|}
dav1d_default_picture_release:
   80|   405k|void dav1d_default_picture_release(Dav1dPicture *const p, void *const cookie) {
   81|   405k|    dav1d_mem_pool_push(cookie, p->allocator_data);
   82|   405k|}
dav1d_picture_free_itut_t35:
   99|    972|void dav1d_picture_free_itut_t35(const uint8_t *const data, void *const user_data) {
  100|    972|    struct itut_t35_ctx_context *itut_t35_ctx = user_data;
  101|       |
  102|  3.05k|    for (size_t i = 0; i < itut_t35_ctx->n_itut_t35; i++)
  ------------------
  |  Branch (102:24): [True: 2.08k, False: 972]
  ------------------
  103|  2.08k|        dav1d_free(itut_t35_ctx->itut_t35[i].payload);
  ------------------
  |  |  135|  2.08k|#define dav1d_free(ptr) free(ptr)
  ------------------
  104|    972|    dav1d_free(itut_t35_ctx->itut_t35);
  ------------------
  |  |  135|    972|#define dav1d_free(ptr) free(ptr)
  ------------------
  105|    972|    dav1d_free(itut_t35_ctx);
  ------------------
  |  |  135|    972|#define dav1d_free(ptr) free(ptr)
  ------------------
  106|    972|}
dav1d_picture_copy_props:
  164|   372k|{
  165|   372k|    dav1d_data_props_copy(&p->m, props);
  166|       |
  167|   372k|    dav1d_ref_dec(&p->content_light_ref);
  168|   372k|    p->content_light_ref = content_light_ref;
  169|   372k|    p->content_light = content_light;
  170|   372k|    if (content_light_ref) dav1d_ref_inc(content_light_ref);
  ------------------
  |  Branch (170:9): [True: 6.62k, False: 366k]
  ------------------
  171|       |
  172|   372k|    dav1d_ref_dec(&p->mastering_display_ref);
  173|   372k|    p->mastering_display_ref = mastering_display_ref;
  174|   372k|    p->mastering_display = mastering_display;
  175|   372k|    if (mastering_display_ref) dav1d_ref_inc(mastering_display_ref);
  ------------------
  |  Branch (175:9): [True: 778, False: 372k]
  ------------------
  176|       |
  177|   372k|    dav1d_ref_dec(&p->itut_t35_ref);
  178|   372k|    p->itut_t35_ref = itut_t35_ref;
  179|   372k|    p->itut_t35 = itut_t35;
  180|   372k|    p->n_itut_t35 = n_itut_t35;
  181|   372k|    if (itut_t35_ref) dav1d_ref_inc(itut_t35_ref);
  ------------------
  |  Branch (181:9): [True: 1.15k, False: 371k]
  ------------------
  182|   372k|}
dav1d_thread_picture_alloc:
  186|   356k|{
  187|   356k|    Dav1dThreadPicture *const p = &f->sr_cur;
  188|       |
  189|   356k|    const int res = picture_alloc(c, &p->p, f->frame_hdr->width[1], f->frame_hdr->height,
  190|   356k|                                  f->seq_hdr, f->seq_hdr_ref,
  191|   356k|                                  f->frame_hdr, f->frame_hdr_ref,
  192|   356k|                                  bpc, &f->tile[0].data.m, &c->allocator,
  193|   356k|                                  (void **) &p->progress);
  194|   356k|    if (res) return res;
  ------------------
  |  Branch (194:9): [True: 0, False: 356k]
  ------------------
  195|       |
  196|       |    // Don't clear these flags from c->frame_flags if the frame is not going to be output.
  197|       |    // This way they will be added to the next visible frame too.
  198|   356k|    const int flags_mask = ((f->frame_hdr->show_frame || c->output_invisible_frames) &&
  ------------------
  |  Branch (198:30): [True: 292k, False: 64.0k]
  |  Branch (198:58): [True: 0, False: 64.0k]
  ------------------
  199|   292k|                            c->max_spatial_id == f->frame_hdr->spatial_id)
  ------------------
  |  Branch (199:29): [True: 281k, False: 10.9k]
  ------------------
  200|   356k|                           ? 0 : (PICTURE_FLAG_NEW_SEQUENCE | PICTURE_FLAG_NEW_OP_PARAMS_INFO);
  201|   356k|    p->flags = c->frame_flags;
  202|   356k|    c->frame_flags &= flags_mask;
  203|       |
  204|   356k|    p->visible = f->frame_hdr->show_frame;
  205|   356k|    p->showable = f->frame_hdr->showable_frame;
  206|       |
  207|   356k|    if (p->visible) {
  ------------------
  |  Branch (207:9): [True: 292k, False: 64.0k]
  ------------------
  208|       |        // Only add HDR10+ and T35 metadata when show frame flag is enabled
  209|   292k|        dav1d_picture_copy_props(&p->p, c->content_light, c->content_light_ref,
  210|   292k|                                 c->mastering_display, c->mastering_display_ref,
  211|   292k|                                 c->itut_t35, c->itut_t35_ref, c->n_itut_t35,
  212|   292k|                                 &f->tile[0].data.m);
  213|       |
  214|       |        // Must be removed from the context after being attached to the frame
  215|   292k|        dav1d_ref_dec(&c->itut_t35_ref);
  216|   292k|        c->itut_t35 = NULL;
  217|   292k|        c->n_itut_t35 = 0;
  218|   292k|    } else {
  219|  64.0k|        dav1d_data_props_copy(&p->p.m, &f->tile[0].data.m);
  220|  64.0k|    }
  221|       |
  222|   356k|    if (c->n_fc > 1) {
  ------------------
  |  Branch (222:9): [True: 356k, False: 0]
  ------------------
  223|   356k|        atomic_init(&p->progress[0], 0);
  224|       |        atomic_init(&p->progress[1], 0);
  225|   356k|    }
  226|   356k|    return res;
  227|   356k|}
dav1d_picture_alloc_copy:
  231|  49.2k|{
  232|  49.2k|    struct pic_ctx_context *const pic_ctx = (struct pic_ctx_context*)src->ref->const_data;
  233|  49.2k|    const int res = picture_alloc(c, dst, w, src->p.h,
  234|  49.2k|                                  src->seq_hdr, src->seq_hdr_ref,
  235|  49.2k|                                  src->frame_hdr, src->frame_hdr_ref,
  236|  49.2k|                                  src->p.bpc, &src->m, &pic_ctx->allocator,
  237|  49.2k|                                  NULL);
  238|  49.2k|    if (res) return res;
  ------------------
  |  Branch (238:9): [True: 0, False: 49.2k]
  ------------------
  239|       |
  240|  49.2k|    dav1d_picture_copy_props(dst, src->content_light, src->content_light_ref,
  241|  49.2k|                             src->mastering_display, src->mastering_display_ref,
  242|  49.2k|                             src->itut_t35, src->itut_t35_ref, src->n_itut_t35,
  243|  49.2k|                             &src->m);
  244|       |
  245|  49.2k|    return 0;
  246|  49.2k|}
dav1d_picture_ref:
  248|  4.07M|void dav1d_picture_ref(Dav1dPicture *const dst, const Dav1dPicture *const src) {
  249|  4.07M|    assert(dst != NULL);
  ------------------
  |  Branch (249:5): [True: 4.07M, False: 0]
  ------------------
  250|  4.07M|    assert(dst->data[0] == NULL);
  ------------------
  |  Branch (250:5): [True: 4.07M, False: 0]
  ------------------
  251|  4.07M|    assert(src != NULL);
  ------------------
  |  Branch (251:5): [True: 4.07M, False: 0]
  ------------------
  252|       |
  253|  4.07M|    if (src->ref) {
  ------------------
  |  Branch (253:9): [True: 4.07M, False: 0]
  ------------------
  254|  4.07M|        assert(src->data[0] != NULL);
  ------------------
  |  Branch (254:9): [True: 4.07M, False: 0]
  ------------------
  255|  4.07M|        dav1d_ref_inc(src->ref);
  256|  4.07M|    }
  257|  4.07M|    if (src->frame_hdr_ref) dav1d_ref_inc(src->frame_hdr_ref);
  ------------------
  |  Branch (257:9): [True: 4.07M, False: 0]
  ------------------
  258|  4.07M|    if (src->seq_hdr_ref) dav1d_ref_inc(src->seq_hdr_ref);
  ------------------
  |  Branch (258:9): [True: 4.07M, False: 0]
  ------------------
  259|  4.07M|    if (src->m.user_data.ref) dav1d_ref_inc(src->m.user_data.ref);
  ------------------
  |  Branch (259:9): [True: 0, False: 4.07M]
  ------------------
  260|  4.07M|    if (src->content_light_ref) dav1d_ref_inc(src->content_light_ref);
  ------------------
  |  Branch (260:9): [True: 44.0k, False: 4.02M]
  ------------------
  261|  4.07M|    if (src->mastering_display_ref) dav1d_ref_inc(src->mastering_display_ref);
  ------------------
  |  Branch (261:9): [True: 3.24k, False: 4.06M]
  ------------------
  262|  4.07M|    if (src->itut_t35_ref) dav1d_ref_inc(src->itut_t35_ref);
  ------------------
  |  Branch (262:9): [True: 5.84k, False: 4.06M]
  ------------------
  263|  4.07M|    *dst = *src;
  264|  4.07M|}
dav1d_picture_move_ref:
  266|   166k|void dav1d_picture_move_ref(Dav1dPicture *const dst, Dav1dPicture *const src) {
  267|   166k|    assert(dst != NULL);
  ------------------
  |  Branch (267:5): [True: 166k, False: 0]
  ------------------
  268|   166k|    assert(dst->data[0] == NULL);
  ------------------
  |  Branch (268:5): [True: 166k, False: 0]
  ------------------
  269|   166k|    assert(src != NULL);
  ------------------
  |  Branch (269:5): [True: 166k, False: 0]
  ------------------
  270|       |
  271|   166k|    if (src->ref)
  ------------------
  |  Branch (271:9): [True: 166k, False: 0]
  ------------------
  272|   166k|        assert(src->data[0] != NULL);
  ------------------
  |  Branch (272:9): [True: 166k, False: 0]
  ------------------
  273|       |
  274|   166k|    *dst = *src;
  275|   166k|    memset(src, 0, sizeof(*src));
  276|   166k|}
dav1d_thread_picture_ref:
  280|  3.75M|{
  281|  3.75M|    dav1d_picture_ref(&dst->p, &src->p);
  282|  3.75M|    dst->visible = src->visible;
  283|  3.75M|    dst->showable = src->showable;
  284|  3.75M|    dst->progress = src->progress;
  285|  3.75M|    dst->flags = src->flags;
  286|  3.75M|}
dav1d_picture_unref_internal:
  299|  4.72M|void dav1d_picture_unref_internal(Dav1dPicture *const p) {
  300|  4.72M|    validate_input(p != NULL);
  ------------------
  |  |   59|  4.72M|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|  4.72M|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 4.72M]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  301|       |
  302|  4.72M|    if (p->ref) {
  ------------------
  |  Branch (302:9): [True: 4.47M, False: 246k]
  ------------------
  303|  4.47M|        validate_input(p->data[0] != NULL);
  ------------------
  |  |   59|  4.47M|#define validate_input(x) validate_input_or_ret(x, )
  |  |  ------------------
  |  |  |  |   52|  4.47M|    if (!(x)) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (52:9): [True: 0, False: 4.47M]
  |  |  |  |  ------------------
  |  |  |  |   53|      0|        debug_print("Input validation check \'%s\' failed in %s!\n", \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|#define debug_print(...) fprintf(stderr, __VA_ARGS__)
  |  |  |  |  ------------------
  |  |  |  |   54|      0|                    #x, __func__); \
  |  |  |  |   55|      0|        debug_abort(); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   39|      0|#define debug_abort abort
  |  |  |  |  ------------------
  |  |  |  |   56|      0|        return r; \
  |  |  |  |   57|      0|    }
  |  |  ------------------
  ------------------
  304|  4.47M|        dav1d_ref_dec(&p->ref);
  305|  4.47M|    }
  306|  4.72M|    dav1d_ref_dec(&p->seq_hdr_ref);
  307|  4.72M|    dav1d_ref_dec(&p->frame_hdr_ref);
  308|  4.72M|    dav1d_ref_dec(&p->m.user_data.ref);
  309|  4.72M|    dav1d_ref_dec(&p->content_light_ref);
  310|  4.72M|    dav1d_ref_dec(&p->mastering_display_ref);
  311|  4.72M|    dav1d_ref_dec(&p->itut_t35_ref);
  312|  4.72M|    memset(p, 0, sizeof(*p));
  313|  4.72M|    dav1d_data_props_set_defaults(&p->m);
  314|  4.72M|}
dav1d_thread_picture_unref:
  316|  4.15M|void dav1d_thread_picture_unref(Dav1dThreadPicture *const p) {
  317|  4.15M|    dav1d_picture_unref_internal(&p->p);
  318|       |
  319|       |    p->progress = NULL;
  320|  4.15M|}
dav1d_picture_get_event_flags:
  322|   176k|enum Dav1dEventFlags dav1d_picture_get_event_flags(const Dav1dThreadPicture *const p) {
  323|   176k|    if (!p->flags)
  ------------------
  |  Branch (323:9): [True: 147k, False: 29.2k]
  ------------------
  324|   147k|        return 0;
  325|       |
  326|  29.2k|    enum Dav1dEventFlags flags = 0;
  327|  29.2k|    if (p->flags & PICTURE_FLAG_NEW_SEQUENCE)
  ------------------
  |  Branch (327:9): [True: 14.9k, False: 14.2k]
  ------------------
  328|  14.9k|       flags |= DAV1D_EVENT_FLAG_NEW_SEQUENCE;
  329|  29.2k|    if (p->flags & PICTURE_FLAG_NEW_OP_PARAMS_INFO)
  ------------------
  |  Branch (329:9): [True: 18, False: 29.2k]
  ------------------
  330|     18|       flags |= DAV1D_EVENT_FLAG_NEW_OP_PARAMS_INFO;
  331|       |
  332|  29.2k|    return flags;
  333|   176k|}
picture.c:picture_alloc:
  117|   405k|{
  118|   405k|    if (p->data[0]) {
  ------------------
  |  Branch (118:9): [True: 0, False: 405k]
  ------------------
  119|      0|        dav1d_log(c, "Picture already allocated!\n");
  ------------------
  |  |   44|      0|#define dav1d_log(...) do { } while(0)
  |  |  ------------------
  |  |  |  Branch (44:37): [Folded, False: 0]
  |  |  ------------------
  ------------------
  120|      0|        return -1;
  121|      0|    }
  122|   405k|    assert(bpc > 0 && bpc <= 16);
  ------------------
  |  Branch (122:5): [True: 405k, False: 0]
  |  Branch (122:5): [True: 405k, False: 0]
  ------------------
  123|       |
  124|   405k|    size_t extra = c->n_fc > 1 ? sizeof(atomic_int) * 2 : 0;
  ------------------
  |  Branch (124:20): [True: 405k, False: 0]
  ------------------
  125|   405k|    struct pic_ctx_context *pic_ctx = dav1d_mem_pool_pop(c->pic_ctx_pool, extra +
  126|   405k|                                                         sizeof(struct pic_ctx_context));
  127|   405k|    if (!pic_ctx)
  ------------------
  |  Branch (127:9): [True: 0, False: 405k]
  ------------------
  128|      0|        return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  129|       |
  130|   405k|    p->p.w = w;
  131|   405k|    p->p.h = h;
  132|   405k|    p->seq_hdr = seq_hdr;
  133|   405k|    p->frame_hdr = frame_hdr;
  134|   405k|    p->p.layout = seq_hdr->layout;
  135|   405k|    p->p.bpc = bpc;
  136|   405k|    dav1d_data_props_set_defaults(&p->m);
  137|   405k|    const int res = p_allocator->alloc_picture_callback(p, p_allocator->cookie);
  138|   405k|    if (res < 0) {
  ------------------
  |  Branch (138:9): [True: 0, False: 405k]
  ------------------
  139|      0|        dav1d_mem_pool_push(c->pic_ctx_pool, pic_ctx);
  140|      0|        return res;
  141|      0|    }
  142|       |
  143|   405k|    pic_ctx->allocator = *p_allocator;
  144|   405k|    pic_ctx->pic = *p;
  145|   405k|    p->ref = dav1d_ref_init(&pic_ctx->ref, pic_ctx, free_buffer, c->pic_ctx_pool, 0);
  146|       |
  147|   405k|    p->seq_hdr_ref = seq_hdr_ref;
  148|   405k|    if (seq_hdr_ref) dav1d_ref_inc(seq_hdr_ref);
  ------------------
  |  Branch (148:9): [True: 405k, False: 0]
  ------------------
  149|       |
  150|   405k|    p->frame_hdr_ref = frame_hdr_ref;
  151|   405k|    if (frame_hdr_ref) dav1d_ref_inc(frame_hdr_ref);
  ------------------
  |  Branch (151:9): [True: 405k, False: 0]
  ------------------
  152|       |
  153|   405k|    if (extra && extra_ptr)
  ------------------
  |  Branch (153:9): [True: 405k, False: 0]
  |  Branch (153:18): [True: 356k, False: 49.2k]
  ------------------
  154|   356k|        *extra_ptr = &pic_ctx->extra_data;
  155|       |
  156|   405k|    return 0;
  157|   405k|}
picture.c:free_buffer:
   91|   405k|static void free_buffer(const uint8_t *const data, void *const user_data) {
   92|   405k|    struct pic_ctx_context *pic_ctx = (struct pic_ctx_context*)data;
   93|       |
   94|   405k|    pic_ctx->allocator.release_picture_callback(&pic_ctx->pic,
   95|   405k|                                                pic_ctx->allocator.cookie);
   96|   405k|    dav1d_mem_pool_push(user_data, pic_ctx);
   97|   405k|}

dav1d_init_qm_tables:
 4684|      1|COLD void dav1d_init_qm_tables(void) {
 4685|       |    // This function is guaranteed to be called only once
 4686|       |
 4687|     16|    for (int i = 0; i < 15; i++)
  ------------------
  |  Branch (4687:21): [True: 15, False: 1]
  ------------------
 4688|     45|        for (int j = 0; j < 2; j++) {
  ------------------
  |  Branch (4688:25): [True: 30, False: 15]
  ------------------
 4689|       |            // note that the w/h in the assignment is inverted, this is on purpose
 4690|       |            // because we store coefficients transposed
 4691|     30|            dav1d_qm_tbl[i][j][RTX_4X8  ] = qm_tbl_8x4[i][j];
 4692|     30|            dav1d_qm_tbl[i][j][RTX_8X4  ] = qm_tbl_4x8[i][j];
 4693|     30|            dav1d_qm_tbl[i][j][RTX_4X16 ] = qm_tbl_16x4[i][j];
 4694|     30|            dav1d_qm_tbl[i][j][RTX_16X4 ] = qm_tbl_4x16[i][j];
 4695|     30|            dav1d_qm_tbl[i][j][RTX_8X16 ] = qm_tbl_16x8[i][j];
 4696|     30|            dav1d_qm_tbl[i][j][RTX_16X8 ] = qm_tbl_8x16[i][j];
 4697|     30|            dav1d_qm_tbl[i][j][RTX_8X32 ] = qm_tbl_32x8[i][j];
 4698|     30|            dav1d_qm_tbl[i][j][RTX_32X8 ] = qm_tbl_8x32[i][j];
 4699|     30|            dav1d_qm_tbl[i][j][RTX_16X32] = qm_tbl_32x16[i][j];
 4700|     30|            dav1d_qm_tbl[i][j][RTX_32X16] = qm_tbl_16x32[i][j];
 4701|       |
 4702|     30|            dav1d_qm_tbl[i][j][ TX_4X4  ] = qm_tbl_4x4[i][j];
 4703|     30|            dav1d_qm_tbl[i][j][ TX_8X8  ] = qm_tbl_8x8[i][j];
 4704|     30|            dav1d_qm_tbl[i][j][ TX_16X16] = qm_tbl_16x16[i][j];
 4705|     30|            dav1d_qm_tbl[i][j][ TX_32X32] = qm_tbl_32x32[i][j];
 4706|       |
 4707|     30|            dav1d_qm_tbl[i][j][ TX_64X64] = dav1d_qm_tbl[i][j][ TX_32X32];
 4708|     30|            dav1d_qm_tbl[i][j][RTX_64X32] = dav1d_qm_tbl[i][j][ TX_32X32];
 4709|     30|            dav1d_qm_tbl[i][j][RTX_64X16] = dav1d_qm_tbl[i][j][RTX_32X16];
 4710|     30|            dav1d_qm_tbl[i][j][RTX_32X64] = dav1d_qm_tbl[i][j][ TX_32X32];
 4711|     30|            dav1d_qm_tbl[i][j][RTX_16X64] = dav1d_qm_tbl[i][j][RTX_16X32];
 4712|     30|        }
 4713|       |
 4714|       |    // dav1d_qm_tbl[15][*][*] == NULL
 4715|      1|}

dav1d_read_coef_blocks_8bpc:
  826|  5.77M|{
  827|  5.77M|    const Dav1dFrameContext *const f = t->f;
  828|  5.77M|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  829|  5.77M|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  830|  5.77M|    const int bx4 = t->bx & 31, by4 = t->by & 31;
  831|  5.77M|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
  832|  5.77M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  833|  5.77M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
  834|  5.77M|    const int cbw4 = (bw4 + ss_hor) >> ss_hor, cbh4 = (bh4 + ss_ver) >> ss_ver;
  835|  5.77M|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (835:28): [True: 4.10M, False: 1.66M]
  ------------------
  836|  4.10M|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (836:29): [True: 3.81M, False: 294k]
  |  Branch (836:45): [True: 146k, False: 147k]
  ------------------
  837|  3.96M|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (837:29): [True: 3.71M, False: 247k]
  |  Branch (837:45): [True: 123k, False: 123k]
  ------------------
  838|       |
  839|  5.77M|    if (b->skip) {
  ------------------
  |  Branch (839:9): [True: 2.99M, False: 2.78M]
  ------------------
  840|  2.99M|        BlockContext *const a = t->a;
  841|  2.99M|        dav1d_memset_pow2[b_dim[2]](&a->lcoef[bx4], 0x40);
  842|  2.99M|        dav1d_memset_pow2[b_dim[3]](&t->l.lcoef[by4], 0x40);
  843|  2.99M|        if (has_chroma) {
  ------------------
  |  Branch (843:13): [True: 1.46M, False: 1.52M]
  ------------------
  844|  1.46M|            dav1d_memset_pow2_fn memset_cw = dav1d_memset_pow2[ulog2(cbw4)];
  845|  1.46M|            dav1d_memset_pow2_fn memset_ch = dav1d_memset_pow2[ulog2(cbh4)];
  846|  1.46M|            memset_cw(&a->ccoef[0][cbx4], 0x40);
  847|  1.46M|            memset_cw(&a->ccoef[1][cbx4], 0x40);
  848|  1.46M|            memset_ch(&t->l.ccoef[0][cby4], 0x40);
  849|  1.46M|            memset_ch(&t->l.ccoef[1][cby4], 0x40);
  850|  1.46M|        }
  851|  2.99M|        return;
  852|  2.99M|    }
  853|       |
  854|  2.78M|    Dav1dTileState *const ts = t->ts;
  855|  2.78M|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
  856|  2.78M|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
  857|  2.78M|    assert(t->frame_thread.pass == 1);
  ------------------
  |  Branch (857:5): [True: 2.78M, False: 18.4E]
  ------------------
  858|  2.78M|    assert(!b->skip);
  ------------------
  |  Branch (858:5): [True: 2.78M, False: 18.4E]
  ------------------
  859|  2.78M|    const TxfmInfo *const uv_t_dim = &dav1d_txfm_dimensions[b->uvtx];
  860|  2.78M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[b->intra ? b->tx : b->max_ytx];
  ------------------
  |  Branch (860:58): [True: 2.10M, False: 680k]
  ------------------
  861|  2.78M|    const uint16_t tx_split[2] = { b->tx_split0, b->tx_split1 };
  862|       |
  863|  5.60M|    for (int init_y = 0; init_y < h4; init_y += 16) {
  ------------------
  |  Branch (863:26): [True: 2.82M, False: 2.78M]
  ------------------
  864|  2.82M|        const int sub_h4 = imin(h4, 16 + init_y);
  865|  5.74M|        for (int init_x = 0; init_x < w4; init_x += 16) {
  ------------------
  |  Branch (865:30): [True: 2.92M, False: 2.82M]
  ------------------
  866|  2.92M|            const int sub_w4 = imin(w4, init_x + 16);
  867|  2.92M|            int y_off = !!init_y, y, x;
  868|  5.98M|            for (y = init_y, t->by += init_y; y < sub_h4;
  ------------------
  |  Branch (868:47): [True: 3.06M, False: 2.92M]
  ------------------
  869|  3.06M|                 y += t_dim->h, t->by += t_dim->h, y_off++)
  870|  3.06M|            {
  871|  3.06M|                int x_off = !!init_x;
  872|  6.76M|                for (x = init_x, t->bx += init_x; x < sub_w4;
  ------------------
  |  Branch (872:51): [True: 3.69M, False: 3.06M]
  ------------------
  873|  3.69M|                     x += t_dim->w, t->bx += t_dim->w, x_off++)
  874|  3.69M|                {
  875|  3.69M|                    if (!b->intra) {
  ------------------
  |  Branch (875:25): [True: 749k, False: 2.94M]
  ------------------
  876|   749k|                        read_coef_tree(t, bs, b, b->max_ytx, 0, tx_split,
  877|   749k|                                       x_off, y_off, NULL);
  878|  2.94M|                    } else {
  879|  2.94M|                        uint8_t cf_ctx = 0x40;
  880|  2.94M|                        enum TxfmType txtp;
  881|  2.94M|                        const int eob =
  882|  2.94M|                            decode_coefs(t, &t->a->lcoef[bx4 + x],
  883|  2.94M|                                         &t->l.lcoef[by4 + y], b->tx, bs, b, 1,
  884|  2.94M|                                         0, ts->frame_thread[1].cf, &txtp, &cf_ctx);
  885|  2.94M|                        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  2.94M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 2.94M]
  |  |  ------------------
  |  |   35|  2.94M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  2.94M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  886|      0|                            printf("Post-y-cf-blk[tx=%d,txtp=%d,eob=%d]: r=%d\n",
  887|      0|                                   b->tx, txtp, eob, ts->msac.rng);
  888|  2.94M|                        *ts->frame_thread[1].cbi++ = eob * (1 << 5) + txtp;
  889|  2.94M|                        ts->frame_thread[1].cf += imin(t_dim->w, 8) * imin(t_dim->h, 8) * 16;
  890|  2.94M|                        dav1d_memset_likely_pow2(&t->a->lcoef[bx4 + x], cf_ctx, imin(t_dim->w, f->bw - t->bx));
  891|  2.94M|                        dav1d_memset_likely_pow2(&t->l.lcoef[by4 + y], cf_ctx, imin(t_dim->h, f->bh - t->by));
  892|  2.94M|                    }
  893|  3.69M|                }
  894|  3.06M|                t->bx -= x;
  895|  3.06M|            }
  896|  2.92M|            t->by -= y;
  897|       |
  898|  2.92M|            if (!has_chroma) continue;
  ------------------
  |  Branch (898:17): [True: 426k, False: 2.49M]
  ------------------
  899|       |
  900|  2.49M|            const int sub_ch4 = imin(ch4, (init_y + 16) >> ss_ver);
  901|  2.49M|            const int sub_cw4 = imin(cw4, (init_x + 16) >> ss_hor);
  902|  7.47M|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (902:30): [True: 4.97M, False: 2.49M]
  ------------------
  903|  10.3M|                for (y = init_y >> ss_ver, t->by += init_y; y < sub_ch4;
  ------------------
  |  Branch (903:61): [True: 5.37M, False: 4.97M]
  ------------------
  904|  5.37M|                     y += uv_t_dim->h, t->by += uv_t_dim->h << ss_ver)
  905|  5.37M|                {
  906|  11.7M|                    for (x = init_x >> ss_hor, t->bx += init_x; x < sub_cw4;
  ------------------
  |  Branch (906:65): [True: 6.41M, False: 5.37M]
  ------------------
  907|  6.41M|                         x += uv_t_dim->w, t->bx += uv_t_dim->w << ss_hor)
  908|  6.41M|                    {
  909|  6.41M|                        uint8_t cf_ctx = 0x40;
  910|  6.41M|                        enum TxfmType txtp;
  911|  6.41M|                        if (!b->intra)
  ------------------
  |  Branch (911:29): [True: 1.08M, False: 5.33M]
  ------------------
  912|  1.08M|                            txtp = t->scratch.txtp_map[(by4 + (y << ss_ver)) * 32 +
  913|  1.08M|                                                        bx4 + (x << ss_hor)];
  914|  6.41M|                        const int eob =
  915|  6.41M|                            decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
  916|  6.41M|                                         &t->l.ccoef[pl][cby4 + y], b->uvtx, bs,
  917|  6.41M|                                         b, b->intra, 1 + pl, ts->frame_thread[1].cf,
  918|  6.41M|                                         &txtp, &cf_ctx);
  919|  6.41M|                        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  6.41M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 6.41M]
  |  |  ------------------
  |  |   35|  6.41M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  6.41M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  920|      0|                            printf("Post-uv-cf-blk[pl=%d,tx=%d,"
  921|      0|                                   "txtp=%d,eob=%d]: r=%d\n",
  922|      0|                                   pl, b->uvtx, txtp, eob, ts->msac.rng);
  923|  6.41M|                        *ts->frame_thread[1].cbi++ = eob * (1 << 5) + txtp;
  924|  6.41M|                        ts->frame_thread[1].cf += uv_t_dim->w * uv_t_dim->h * 16;
  925|  6.41M|                        int ctw = imin(uv_t_dim->w, (f->bw - t->bx + ss_hor) >> ss_hor);
  926|  6.41M|                        int cth = imin(uv_t_dim->h, (f->bh - t->by + ss_ver) >> ss_ver);
  927|  6.41M|                        dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
  928|  6.41M|                        dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
  929|  6.41M|                    }
  930|  5.37M|                    t->bx -= x << ss_hor;
  931|  5.37M|                }
  932|  4.97M|                t->by -= y << ss_ver;
  933|  4.97M|            }
  934|  2.49M|        }
  935|  2.82M|    }
  936|  2.78M|}
dav1d_recon_b_intra_8bpc:
 1179|   847k|{
 1180|   847k|    Dav1dTileState *const ts = t->ts;
 1181|   847k|    const Dav1dFrameContext *const f = t->f;
 1182|   847k|    const Dav1dDSPContext *const dsp = f->dsp;
 1183|   847k|    const int bx4 = t->bx & 31, by4 = t->by & 31;
 1184|   847k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1185|   847k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 1186|   847k|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
 1187|   847k|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
 1188|   847k|    const int bw4 = b_dim[0], bh4 = b_dim[1];
 1189|   847k|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
 1190|   847k|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
 1191|   847k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (1191:28): [True: 605k, False: 241k]
  ------------------
 1192|   605k|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (1192:29): [True: 497k, False: 108k]
  |  Branch (1192:45): [True: 54.2k, False: 54.0k]
  ------------------
 1193|   552k|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (1193:29): [True: 460k, False: 91.9k]
  |  Branch (1193:45): [True: 46.0k, False: 45.9k]
  ------------------
 1194|   847k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[b->tx];
 1195|   847k|    const TxfmInfo *const uv_t_dim = &dav1d_txfm_dimensions[b->uvtx];
 1196|       |
 1197|       |    // coefficient coding
 1198|   847k|    pixel *const edge = bitfn(t->scratch.edge) + 128;
  ------------------
  |  |   51|   847k|#define bitfn(x) x##_8bpc
  ------------------
 1199|   847k|    const int cbw4 = (bw4 + ss_hor) >> ss_hor, cbh4 = (bh4 + ss_ver) >> ss_ver;
 1200|       |
 1201|   847k|    const int intra_edge_filter_flag = f->seq_hdr->intra_edge_filter << 10;
 1202|       |
 1203|  1.77M|    for (int init_y = 0; init_y < h4; init_y += 16) {
  ------------------
  |  Branch (1203:26): [True: 924k, False: 848k]
  ------------------
 1204|   924k|        const int sub_h4 = imin(h4, 16 + init_y);
 1205|   924k|        const int sub_ch4 = imin(ch4, (init_y + 16) >> ss_ver);
 1206|  1.99M|        for (int init_x = 0; init_x < w4; init_x += 16) {
  ------------------
  |  Branch (1206:30): [True: 1.07M, False: 925k]
  ------------------
 1207|  1.07M|            if (b->pal_sz[0]) {
  ------------------
  |  Branch (1207:17): [True: 13.9k, False: 1.05M]
  ------------------
 1208|  13.9k|                pixel *dst = ((pixel *) f->cur.data[0]) +
 1209|  13.9k|                             4 * (t->by * PXSTRIDE(f->cur.stride[0]) + t->bx);
  ------------------
  |  |   53|  13.9k|#define PXSTRIDE(x) (x)
  ------------------
 1210|  13.9k|                const uint8_t *pal_idx;
 1211|  13.9k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (1211:21): [True: 13.9k, False: 0]
  ------------------
 1212|  13.9k|                    const int p = t->frame_thread.pass & 1;
 1213|  13.9k|                    assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1213:21): [True: 13.9k, False: 0]
  ------------------
 1214|  13.9k|                    pal_idx = ts->frame_thread[p].pal_idx;
 1215|  13.9k|                    ts->frame_thread[p].pal_idx += bw4 * bh4 * 8;
 1216|  13.9k|                } else {
 1217|      0|                    pal_idx = t->scratch.pal_idx_y;
 1218|      0|                }
 1219|  13.9k|                const pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (1219:42): [True: 13.9k, False: 0]
  ------------------
 1220|  13.9k|                    f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 1221|  13.9k|                                        ((t->bx >> 1) + (t->by & 1))][0] :
 1222|  13.9k|                    bytefn(t->scratch.pal)[0];
  ------------------
  |  |   87|      0|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  13.9k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1223|  13.9k|                f->dsp->ipred.pal_pred(dst, f->cur.stride[0], pal,
 1224|  13.9k|                                       pal_idx, bw4 * 4, bh4 * 4);
 1225|  13.9k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|  13.9k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 13.9k]
  |  |  ------------------
  |  |   35|  13.9k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  13.9k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1226|      0|                    hex_dump(dst, PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
 1227|      0|                             bw4 * 4, bh4 * 4, "y-pal-pred");
 1228|  13.9k|            }
 1229|       |
 1230|  1.07M|            const int intra_flags = (sm_flag(t->a, bx4) |
 1231|  1.07M|                                     sm_flag(&t->l, by4) |
 1232|  1.07M|                                     intra_edge_filter_flag);
 1233|  1.07M|            const int sb_has_tr = init_x + 16 < w4 ? 1 : init_y ? 0 :
  ------------------
  |  Branch (1233:35): [True: 148k, False: 924k]
  |  Branch (1233:58): [True: 76.8k, False: 847k]
  ------------------
 1234|   924k|                              intra_edge_flags & EDGE_I444_TOP_HAS_RIGHT;
 1235|  1.07M|            const int sb_has_bl = init_x ? 0 : init_y + 16 < h4 ? 1 :
  ------------------
  |  Branch (1235:35): [True: 148k, False: 924k]
  |  Branch (1235:48): [True: 76.8k, False: 847k]
  ------------------
 1236|   924k|                              intra_edge_flags & EDGE_I444_LEFT_HAS_BOTTOM;
 1237|  1.07M|            int y, x;
 1238|  1.07M|            const int sub_w4 = imin(w4, init_x + 16);
 1239|  2.76M|            for (y = init_y, t->by += init_y; y < sub_h4;
  ------------------
  |  Branch (1239:47): [True: 1.69M, False: 1.07M]
  ------------------
 1240|  1.69M|                 y += t_dim->h, t->by += t_dim->h)
 1241|  1.69M|            {
 1242|  1.69M|                pixel *dst = ((pixel *) f->cur.data[0]) +
 1243|  1.69M|                               4 * (t->by * PXSTRIDE(f->cur.stride[0]) +
  ------------------
  |  |   53|  1.69M|#define PXSTRIDE(x) (x)
  ------------------
 1244|  1.69M|                                    t->bx + init_x);
 1245|  8.69M|                for (x = init_x, t->bx += init_x; x < sub_w4;
  ------------------
  |  Branch (1245:51): [True: 6.99M, False: 1.69M]
  ------------------
 1246|  6.99M|                     x += t_dim->w, t->bx += t_dim->w)
 1247|  6.99M|                {
 1248|  6.99M|                    if (b->pal_sz[0]) goto skip_y_pred;
  ------------------
  |  Branch (1248:25): [True: 15.1k, False: 6.97M]
  ------------------
 1249|       |
 1250|  6.97M|                    int angle = b->y_angle;
 1251|  6.97M|                    const enum EdgeFlags edge_flags =
 1252|  6.97M|                        (((y > init_y || !sb_has_tr) && (x + t_dim->w >= sub_w4)) ?
  ------------------
  |  Branch (1252:28): [True: 5.46M, False: 1.51M]
  |  Branch (1252:42): [True: 409k, False: 1.10M]
  |  Branch (1252:57): [True: 923k, False: 4.95M]
  ------------------
 1253|  6.05M|                             0 : EDGE_I444_TOP_HAS_RIGHT) |
 1254|  6.97M|                        ((x > init_x || (!sb_has_bl && y + t_dim->h >= sub_h4)) ?
  ------------------
  |  Branch (1254:27): [True: 5.29M, False: 1.68M]
  |  Branch (1254:42): [True: 1.20M, False: 476k]
  |  Branch (1254:56): [True: 747k, False: 457k]
  ------------------
 1255|  6.04M|                             0 : EDGE_I444_LEFT_HAS_BOTTOM);
 1256|  6.97M|                    const pixel *top_sb_edge = NULL;
 1257|  6.97M|                    if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1257:25): [True: 629k, False: 6.34M]
  ------------------
 1258|   629k|                        top_sb_edge = f->ipred_edge[0];
 1259|   629k|                        const int sby = t->by >> f->sb_shift;
 1260|   629k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1261|   629k|                    }
 1262|  6.97M|                    const enum IntraPredMode m =
 1263|  6.97M|                        bytefn(dav1d_prepare_intra_edges)(t->bx,
  ------------------
  |  |   87|  6.97M|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  6.97M|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1264|  6.97M|                                                          t->bx > ts->tiling.col_start,
 1265|  6.97M|                                                          t->by,
 1266|  6.97M|                                                          t->by > ts->tiling.row_start,
 1267|  6.97M|                                                          ts->tiling.col_end,
 1268|  6.97M|                                                          ts->tiling.row_end,
 1269|  6.97M|                                                          edge_flags, dst,
 1270|  6.97M|                                                          f->cur.stride[0], top_sb_edge,
 1271|  6.97M|                                                          b->y_mode, &angle,
 1272|  6.97M|                                                          t_dim->w, t_dim->h,
 1273|  6.97M|                                                          f->seq_hdr->intra_edge_filter,
 1274|  6.97M|                                                          edge HIGHBD_CALL_SUFFIX);
 1275|  6.97M|                    dsp->ipred.intra_pred[m](dst, f->cur.stride[0], edge,
 1276|  6.97M|                                             t_dim->w * 4, t_dim->h * 4,
 1277|  6.97M|                                             angle | intra_flags,
 1278|  6.97M|                                             4 * f->bw - 4 * t->bx,
 1279|  6.97M|                                             4 * f->bh - 4 * t->by
 1280|  6.97M|                                             HIGHBD_CALL_SUFFIX);
 1281|       |
 1282|  6.97M|                    if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  6.97M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 6.97M]
  |  |  ------------------
  |  |   35|  6.97M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  6.97M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                  if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1283|      0|                        hex_dump(edge - t_dim->h * 4, t_dim->h * 4,
 1284|      0|                                 t_dim->h * 4, 2, "l");
 1285|      0|                        hex_dump(edge, 0, 1, 1, "tl");
 1286|      0|                        hex_dump(edge + 1, t_dim->w * 4,
 1287|      0|                                 t_dim->w * 4, 2, "t");
 1288|      0|                        hex_dump(dst, f->cur.stride[0],
 1289|      0|                                 t_dim->w * 4, t_dim->h * 4, "y-intra-pred");
 1290|      0|                    }
 1291|       |
 1292|  6.99M|                skip_y_pred: {}
 1293|  6.99M|                    if (!b->skip) {
  ------------------
  |  Branch (1293:25): [True: 688k, False: 6.30M]
  ------------------
 1294|   688k|                        coef *cf;
 1295|   688k|                        int eob;
 1296|   688k|                        enum TxfmType txtp;
 1297|   688k|                        if (t->frame_thread.pass) {
  ------------------
  |  Branch (1297:29): [True: 688k, False: 18.4E]
  ------------------
 1298|   688k|                            const int p = t->frame_thread.pass & 1;
 1299|   688k|                            const int cbi = *ts->frame_thread[p].cbi++;
 1300|   688k|                            cf = ts->frame_thread[p].cf;
 1301|   688k|                            ts->frame_thread[p].cf += imin(t_dim->w, 8) * imin(t_dim->h, 8) * 16;
 1302|   688k|                            eob  = cbi >> 5;
 1303|   688k|                            txtp = cbi & 0x1f;
 1304|  18.4E|                        } else {
 1305|  18.4E|                            uint8_t cf_ctx;
 1306|  18.4E|                            cf = bitfn(t->cf);
  ------------------
  |  |   51|  18.4E|#define bitfn(x) x##_8bpc
  ------------------
 1307|  18.4E|                            eob = decode_coefs(t, &t->a->lcoef[bx4 + x],
 1308|  18.4E|                                               &t->l.lcoef[by4 + y], b->tx, bs,
 1309|  18.4E|                                               b, 1, 0, cf, &txtp, &cf_ctx);
 1310|  18.4E|                            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  18.4E|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 18.4E]
  |  |  ------------------
  |  |   35|  18.4E|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  18.4E|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1311|      0|                                printf("Post-y-cf-blk[tx=%d,txtp=%d,eob=%d]: r=%d\n",
 1312|      0|                                       b->tx, txtp, eob, ts->msac.rng);
 1313|  18.4E|                            dav1d_memset_likely_pow2(&t->a->lcoef[bx4 + x], cf_ctx, imin(t_dim->w, f->bw - t->bx));
 1314|  18.4E|                            dav1d_memset_likely_pow2(&t->l.lcoef[by4 + y], cf_ctx, imin(t_dim->h, f->bh - t->by));
 1315|  18.4E|                        }
 1316|   688k|                        if (eob >= 0) {
  ------------------
  |  Branch (1316:29): [True: 503k, False: 184k]
  ------------------
 1317|   503k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   503k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 503k]
  |  |  ------------------
  |  |   35|   503k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   503k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1318|      0|                                coef_dump(cf, imin(t_dim->h, 8) * 4,
 1319|      0|                                          imin(t_dim->w, 8) * 4, 3, "dq");
 1320|   503k|                            dsp->itx.itxfm_add[b->tx]
 1321|   503k|                                              [txtp](dst,
 1322|   503k|                                                     f->cur.stride[0],
 1323|   503k|                                                     cf, eob HIGHBD_CALL_SUFFIX);
 1324|   503k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   503k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 503k]
  |  |  ------------------
  |  |   35|   503k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   503k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1325|      0|                                hex_dump(dst, f->cur.stride[0],
 1326|      0|                                         t_dim->w * 4, t_dim->h * 4, "recon");
 1327|   503k|                        }
 1328|  6.30M|                    } else if (!t->frame_thread.pass) {
  ------------------
  |  Branch (1328:32): [True: 0, False: 6.30M]
  ------------------
 1329|      0|                        dav1d_memset_pow2[t_dim->lw](&t->a->lcoef[bx4 + x], 0x40);
 1330|      0|                        dav1d_memset_pow2[t_dim->lh](&t->l.lcoef[by4 + y], 0x40);
 1331|      0|                    }
 1332|  6.99M|                    dst += 4 * t_dim->w;
 1333|  6.99M|                }
 1334|  1.69M|                t->bx -= x;
 1335|  1.69M|            }
 1336|  1.07M|            t->by -= y;
 1337|       |
 1338|  1.07M|            if (!has_chroma) continue;
  ------------------
  |  Branch (1338:17): [True: 408k, False: 665k]
  ------------------
 1339|       |
 1340|   665k|            const ptrdiff_t stride = f->cur.stride[1];
 1341|       |
 1342|   665k|            if (b->uv_mode == CFL_PRED) {
  ------------------
  |  Branch (1342:17): [True: 46.3k, False: 619k]
  ------------------
 1343|  46.3k|                assert(!init_x && !init_y);
  ------------------
  |  Branch (1343:17): [True: 46.3k, False: 0]
  |  Branch (1343:17): [True: 46.3k, False: 0]
  ------------------
 1344|       |
 1345|  46.3k|                int16_t *const ac = t->scratch.ac;
 1346|  46.3k|                pixel *y_src = ((pixel *) f->cur.data[0]) + 4 * (t->bx & ~ss_hor) +
 1347|  46.3k|                                 4 * (t->by & ~ss_ver) * PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|  46.3k|#define PXSTRIDE(x) (x)
  ------------------
 1348|  46.3k|                const ptrdiff_t uv_off = 4 * ((t->bx >> ss_hor) +
 1349|  46.3k|                                              (t->by >> ss_ver) * PXSTRIDE(stride));
  ------------------
  |  |   53|  46.3k|#define PXSTRIDE(x) (x)
  ------------------
 1350|  46.3k|                pixel *const uv_dst[2] = { ((pixel *) f->cur.data[1]) + uv_off,
 1351|  46.3k|                                           ((pixel *) f->cur.data[2]) + uv_off };
 1352|       |
 1353|  46.3k|                const int furthest_r =
 1354|  46.3k|                    ((cw4 << ss_hor) + t_dim->w - 1) & ~(t_dim->w - 1);
 1355|  46.3k|                const int furthest_b =
 1356|  46.3k|                    ((ch4 << ss_ver) + t_dim->h - 1) & ~(t_dim->h - 1);
 1357|  46.3k|                dsp->ipred.cfl_ac[f->cur.p.layout - 1](ac, y_src, f->cur.stride[0],
 1358|  46.3k|                                                         cbw4 - (furthest_r >> ss_hor),
 1359|  46.3k|                                                         cbh4 - (furthest_b >> ss_ver),
 1360|  46.3k|                                                         cbw4 * 4, cbh4 * 4);
 1361|   139k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1361:34): [True: 92.7k, False: 46.3k]
  ------------------
 1362|  92.7k|                    if (!b->cfl_alpha[pl]) continue;
  ------------------
  |  Branch (1362:25): [True: 18.0k, False: 74.6k]
  ------------------
 1363|  74.6k|                    int angle = 0;
 1364|  74.6k|                    const pixel *top_sb_edge = NULL;
 1365|  74.6k|                    if (!((t->by & ~ss_ver) & (f->sb_step - 1))) {
  ------------------
  |  Branch (1365:25): [True: 10.4k, False: 64.1k]
  ------------------
 1366|  10.4k|                        top_sb_edge = f->ipred_edge[pl + 1];
 1367|  10.4k|                        const int sby = t->by >> f->sb_shift;
 1368|  10.4k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1369|  10.4k|                    }
 1370|  74.6k|                    const int xpos = t->bx >> ss_hor, ypos = t->by >> ss_ver;
 1371|  74.6k|                    const int xstart = ts->tiling.col_start >> ss_hor;
 1372|  74.6k|                    const int ystart = ts->tiling.row_start >> ss_ver;
 1373|  74.6k|                    const enum IntraPredMode m =
 1374|  74.6k|                        bytefn(dav1d_prepare_intra_edges)(xpos, xpos > xstart,
  ------------------
  |  |   87|  74.6k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  74.6k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1375|  74.6k|                                                          ypos, ypos > ystart,
 1376|  74.6k|                                                          ts->tiling.col_end >> ss_hor,
 1377|  74.6k|                                                          ts->tiling.row_end >> ss_ver,
 1378|  74.6k|                                                          0, uv_dst[pl], stride,
 1379|  74.6k|                                                          top_sb_edge, DC_PRED, &angle,
 1380|  74.6k|                                                          uv_t_dim->w, uv_t_dim->h, 0,
 1381|  74.6k|                                                          edge HIGHBD_CALL_SUFFIX);
 1382|  74.6k|                    dsp->ipred.cfl_pred[m](uv_dst[pl], stride, edge,
 1383|  74.6k|                                           uv_t_dim->w * 4,
 1384|  74.6k|                                           uv_t_dim->h * 4,
 1385|  74.6k|                                           ac, b->cfl_alpha[pl]
 1386|  74.6k|                                           HIGHBD_CALL_SUFFIX);
 1387|  74.6k|                }
 1388|  46.3k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  46.3k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 46.3k]
  |  |  ------------------
  |  |   35|  46.3k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  46.3k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1389|      0|                    ac_dump(ac, 4*cbw4, 4*cbh4, "ac");
 1390|      0|                    hex_dump(uv_dst[0], stride, cbw4 * 4, cbh4 * 4, "u-cfl-pred");
 1391|      0|                    hex_dump(uv_dst[1], stride, cbw4 * 4, cbh4 * 4, "v-cfl-pred");
 1392|      0|                }
 1393|   619k|            } else if (b->pal_sz[1]) {
  ------------------
  |  Branch (1393:24): [True: 5.78k, False: 613k]
  ------------------
 1394|  5.78k|                const ptrdiff_t uv_dstoff = 4 * ((t->bx >> ss_hor) +
 1395|  5.78k|                                              (t->by >> ss_ver) * PXSTRIDE(f->cur.stride[1]));
  ------------------
  |  |   53|  5.78k|#define PXSTRIDE(x) (x)
  ------------------
 1396|  5.78k|                const pixel (*pal)[8];
 1397|  5.78k|                const uint8_t *pal_idx;
 1398|  5.78k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (1398:21): [True: 5.78k, False: 0]
  ------------------
 1399|  5.78k|                    const int p = t->frame_thread.pass & 1;
 1400|  5.78k|                    assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1400:21): [True: 5.78k, False: 0]
  ------------------
 1401|  5.78k|                    pal = f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 1402|  5.78k|                                              ((t->bx >> 1) + (t->by & 1))];
 1403|  5.78k|                    pal_idx = ts->frame_thread[p].pal_idx;
 1404|  5.78k|                    ts->frame_thread[p].pal_idx += cbw4 * cbh4 * 8;
 1405|  5.78k|                } else {
 1406|      0|                    pal = bytefn(t->scratch.pal);
  ------------------
  |  |   87|      0|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|      0|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1407|      0|                    pal_idx = t->scratch.pal_idx_uv;
 1408|      0|                }
 1409|       |
 1410|  5.78k|                f->dsp->ipred.pal_pred(((pixel *) f->cur.data[1]) + uv_dstoff,
 1411|  5.78k|                                       f->cur.stride[1], pal[1],
 1412|  5.78k|                                       pal_idx, cbw4 * 4, cbh4 * 4);
 1413|  5.78k|                f->dsp->ipred.pal_pred(((pixel *) f->cur.data[2]) + uv_dstoff,
 1414|  5.78k|                                       f->cur.stride[1], pal[2],
 1415|  5.78k|                                       pal_idx, cbw4 * 4, cbh4 * 4);
 1416|  5.78k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  5.78k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 5.78k]
  |  |  ------------------
  |  |   35|  5.78k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  5.78k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1417|      0|                    hex_dump(((pixel *) f->cur.data[1]) + uv_dstoff,
 1418|      0|                             PXSTRIDE(f->cur.stride[1]),
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
 1419|      0|                             cbw4 * 4, cbh4 * 4, "u-pal-pred");
 1420|      0|                    hex_dump(((pixel *) f->cur.data[2]) + uv_dstoff,
 1421|      0|                             PXSTRIDE(f->cur.stride[1]),
  ------------------
  |  |   53|      0|#define PXSTRIDE(x) (x)
  ------------------
 1422|      0|                             cbw4 * 4, cbh4 * 4, "v-pal-pred");
 1423|      0|                }
 1424|  5.78k|            }
 1425|       |
 1426|   665k|            const int sm_uv_fl = sm_uv_flag(t->a, cbx4) |
 1427|   665k|                                 sm_uv_flag(&t->l, cby4);
 1428|   665k|            const int uv_sb_has_tr =
 1429|   665k|                ((init_x + 16) >> ss_hor) < cw4 ? 1 : init_y ? 0 :
  ------------------
  |  Branch (1429:17): [True: 103k, False: 561k]
  |  Branch (1429:55): [True: 54.7k, False: 507k]
  ------------------
 1430|   561k|                intra_edge_flags & (EDGE_I420_TOP_HAS_RIGHT >> (f->cur.p.layout - 1));
 1431|   665k|            const int uv_sb_has_bl =
 1432|   665k|                init_x ? 0 : ((init_y + 16) >> ss_ver) < ch4 ? 1 :
  ------------------
  |  Branch (1432:17): [True: 103k, False: 561k]
  |  Branch (1432:30): [True: 54.7k, False: 507k]
  ------------------
 1433|   561k|                intra_edge_flags & (EDGE_I420_LEFT_HAS_BOTTOM >> (f->cur.p.layout - 1));
 1434|   665k|            const int sub_cw4 = imin(cw4, (init_x + 16) >> ss_hor);
 1435|  1.99M|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1435:30): [True: 1.32M, False: 665k]
  ------------------
 1436|  3.55M|                for (y = init_y >> ss_ver, t->by += init_y; y < sub_ch4;
  ------------------
  |  Branch (1436:61): [True: 2.22M, False: 1.32M]
  ------------------
 1437|  2.22M|                     y += uv_t_dim->h, t->by += uv_t_dim->h << ss_ver)
 1438|  2.22M|                {
 1439|  2.22M|                    pixel *dst = ((pixel *) f->cur.data[1 + pl]) +
 1440|  2.22M|                                   4 * ((t->by >> ss_ver) * PXSTRIDE(stride) +
  ------------------
  |  |   53|  2.22M|#define PXSTRIDE(x) (x)
  ------------------
 1441|  2.22M|                                        ((t->bx + init_x) >> ss_hor));
 1442|  8.78M|                    for (x = init_x >> ss_hor, t->bx += init_x; x < sub_cw4;
  ------------------
  |  Branch (1442:65): [True: 6.56M, False: 2.22M]
  ------------------
 1443|  6.56M|                         x += uv_t_dim->w, t->bx += uv_t_dim->w << ss_hor)
 1444|  6.56M|                    {
 1445|  6.56M|                        if ((b->uv_mode == CFL_PRED && b->cfl_alpha[pl]) ||
  ------------------
  |  Branch (1445:30): [True: 92.7k, False: 6.47M]
  |  Branch (1445:56): [True: 74.6k, False: 18.0k]
  ------------------
 1446|  6.49M|                            b->pal_sz[1])
  ------------------
  |  Branch (1446:29): [True: 12.4k, False: 6.47M]
  ------------------
 1447|  87.1k|                        {
 1448|  87.1k|                            goto skip_uv_pred;
 1449|  87.1k|                        }
 1450|       |
 1451|  6.47M|                        int angle = b->uv_angle;
 1452|       |                        // this probably looks weird because we're using
 1453|       |                        // luma flags in a chroma loop, but that's because
 1454|       |                        // prepare_intra_edges() expects luma flags as input
 1455|  6.47M|                        const enum EdgeFlags edge_flags =
 1456|  6.47M|                            (((y > (init_y >> ss_ver) || !uv_sb_has_tr) &&
  ------------------
  |  Branch (1456:32): [True: 4.83M, False: 1.64M]
  |  Branch (1456:58): [True: 491k, False: 1.15M]
  ------------------
 1457|  5.32M|                              (x + uv_t_dim->w >= sub_cw4)) ?
  ------------------
  |  Branch (1457:31): [True: 1.32M, False: 4.00M]
  ------------------
 1458|  5.15M|                                 0 : EDGE_I444_TOP_HAS_RIGHT) |
 1459|  6.47M|                            ((x > (init_x >> ss_hor) ||
  ------------------
  |  Branch (1459:31): [True: 4.33M, False: 2.13M]
  ------------------
 1460|  2.13M|                              (!uv_sb_has_bl && y + uv_t_dim->h >= sub_ch4)) ?
  ------------------
  |  Branch (1460:32): [True: 1.45M, False: 679k]
  |  Branch (1460:49): [True: 798k, False: 660k]
  ------------------
 1461|  5.14M|                                 0 : EDGE_I444_LEFT_HAS_BOTTOM);
 1462|  6.47M|                        const pixel *top_sb_edge = NULL;
 1463|  6.47M|                        if (!((t->by & ~ss_ver) & (f->sb_step - 1))) {
  ------------------
  |  Branch (1463:29): [True: 573k, False: 5.90M]
  ------------------
 1464|   573k|                            top_sb_edge = f->ipred_edge[1 + pl];
 1465|   573k|                            const int sby = t->by >> f->sb_shift;
 1466|   573k|                            top_sb_edge += f->sb128w * 128 * (sby - 1);
 1467|   573k|                        }
 1468|  6.47M|                        const enum IntraPredMode uv_mode =
 1469|  6.47M|                             b->uv_mode == CFL_PRED ? DC_PRED : b->uv_mode;
  ------------------
  |  Branch (1469:30): [True: 18.0k, False: 6.45M]
  ------------------
 1470|  6.47M|                        const int xpos = t->bx >> ss_hor, ypos = t->by >> ss_ver;
 1471|  6.47M|                        const int xstart = ts->tiling.col_start >> ss_hor;
 1472|  6.47M|                        const int ystart = ts->tiling.row_start >> ss_ver;
 1473|  6.47M|                        const enum IntraPredMode m =
 1474|  6.47M|                            bytefn(dav1d_prepare_intra_edges)(xpos, xpos > xstart,
  ------------------
  |  |   87|  6.47M|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  6.47M|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1475|  6.47M|                                                              ypos, ypos > ystart,
 1476|  6.47M|                                                              ts->tiling.col_end >> ss_hor,
 1477|  6.47M|                                                              ts->tiling.row_end >> ss_ver,
 1478|  6.47M|                                                              edge_flags, dst, stride,
 1479|  6.47M|                                                              top_sb_edge, uv_mode,
 1480|  6.47M|                                                              &angle, uv_t_dim->w,
 1481|  6.47M|                                                              uv_t_dim->h,
 1482|  6.47M|                                                              f->seq_hdr->intra_edge_filter,
 1483|  6.47M|                                                              edge HIGHBD_CALL_SUFFIX);
 1484|  6.47M|                        angle |= intra_edge_filter_flag;
 1485|  6.47M|                        dsp->ipred.intra_pred[m](dst, stride, edge,
 1486|  6.47M|                                                 uv_t_dim->w * 4,
 1487|  6.47M|                                                 uv_t_dim->h * 4,
 1488|  6.47M|                                                 angle | sm_uv_fl,
 1489|  6.47M|                                                 (4 * f->bw + ss_hor -
 1490|  6.47M|                                                  4 * (t->bx & ~ss_hor)) >> ss_hor,
 1491|  6.47M|                                                 (4 * f->bh + ss_ver -
 1492|  6.47M|                                                  4 * (t->by & ~ss_ver)) >> ss_ver
 1493|  6.47M|                                                 HIGHBD_CALL_SUFFIX);
 1494|  6.47M|                        if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  6.47M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 6.47M]
  |  |  ------------------
  |  |   35|  6.47M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  6.47M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                      if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1495|      0|                            hex_dump(edge - uv_t_dim->h * 4, uv_t_dim->h * 4,
 1496|      0|                                     uv_t_dim->h * 4, 2, "l");
 1497|      0|                            hex_dump(edge, 0, 1, 1, "tl");
 1498|      0|                            hex_dump(edge + 1, uv_t_dim->w * 4,
 1499|      0|                                     uv_t_dim->w * 4, 2, "t");
 1500|      0|                            hex_dump(dst, stride, uv_t_dim->w * 4,
 1501|      0|                                     uv_t_dim->h * 4, pl ? "v-intra-pred" : "u-intra-pred");
  ------------------
  |  Branch (1501:55): [True: 0, False: 0]
  ------------------
 1502|      0|                        }
 1503|       |
 1504|  6.56M|                    skip_uv_pred: {}
 1505|  6.56M|                        if (!b->skip) {
  ------------------
  |  Branch (1505:29): [True: 741k, False: 5.82M]
  ------------------
 1506|   741k|                            enum TxfmType txtp;
 1507|   741k|                            int eob;
 1508|   741k|                            coef *cf;
 1509|   741k|                            if (t->frame_thread.pass) {
  ------------------
  |  Branch (1509:33): [True: 741k, False: 18.4E]
  ------------------
 1510|   741k|                                const int p = t->frame_thread.pass & 1;
 1511|   741k|                                const int cbi = *ts->frame_thread[p].cbi++;
 1512|   741k|                                cf = ts->frame_thread[p].cf;
 1513|   741k|                                ts->frame_thread[p].cf += uv_t_dim->w * uv_t_dim->h * 16;
 1514|   741k|                                eob  = cbi >> 5;
 1515|   741k|                                txtp = cbi & 0x1f;
 1516|  18.4E|                            } else {
 1517|  18.4E|                                uint8_t cf_ctx;
 1518|  18.4E|                                cf = bitfn(t->cf);
  ------------------
  |  |   51|  18.4E|#define bitfn(x) x##_8bpc
  ------------------
 1519|  18.4E|                                eob = decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
 1520|  18.4E|                                                   &t->l.ccoef[pl][cby4 + y],
 1521|  18.4E|                                                   b->uvtx, bs, b, 1, 1 + pl, cf,
 1522|  18.4E|                                                   &txtp, &cf_ctx);
 1523|  18.4E|                                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  18.4E|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 18.4E]
  |  |  ------------------
  |  |   35|  18.4E|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  18.4E|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1524|      0|                                    printf("Post-uv-cf-blk[pl=%d,tx=%d,"
 1525|      0|                                           "txtp=%d,eob=%d]: r=%d [x=%d,cbx4=%d]\n",
 1526|      0|                                           pl, b->uvtx, txtp, eob, ts->msac.rng, x, cbx4);
 1527|  18.4E|                                int ctw = imin(uv_t_dim->w, (f->bw - t->bx + ss_hor) >> ss_hor);
 1528|  18.4E|                                int cth = imin(uv_t_dim->h, (f->bh - t->by + ss_ver) >> ss_ver);
 1529|  18.4E|                                dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
 1530|  18.4E|                                dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
 1531|  18.4E|                            }
 1532|   741k|                            if (eob >= 0) {
  ------------------
  |  Branch (1532:33): [True: 281k, False: 459k]
  ------------------
 1533|   281k|                                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   281k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 281k]
  |  |  ------------------
  |  |   35|   281k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   281k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1534|      0|                                    coef_dump(cf, uv_t_dim->h * 4,
 1535|      0|                                              uv_t_dim->w * 4, 3, "dq");
 1536|   281k|                                dsp->itx.itxfm_add[b->uvtx]
 1537|   281k|                                                  [txtp](dst, stride,
 1538|   281k|                                                         cf, eob HIGHBD_CALL_SUFFIX);
 1539|   281k|                                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   281k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 281k]
  |  |  ------------------
  |  |   35|   281k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   281k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1540|      0|                                    hex_dump(dst, stride, uv_t_dim->w * 4,
 1541|      0|                                             uv_t_dim->h * 4, "recon");
 1542|   281k|                            }
 1543|  5.82M|                        } else if (!t->frame_thread.pass) {
  ------------------
  |  Branch (1543:36): [True: 0, False: 5.82M]
  ------------------
 1544|      0|                            dav1d_memset_pow2[uv_t_dim->lw](&t->a->ccoef[pl][cbx4 + x], 0x40);
 1545|      0|                            dav1d_memset_pow2[uv_t_dim->lh](&t->l.ccoef[pl][cby4 + y], 0x40);
 1546|      0|                        }
 1547|  6.56M|                        dst += uv_t_dim->w * 4;
 1548|  6.56M|                    }
 1549|  2.22M|                    t->bx -= x << ss_hor;
 1550|  2.22M|                }
 1551|  1.32M|                t->by -= y << ss_ver;
 1552|  1.32M|            }
 1553|   665k|        }
 1554|   924k|    }
 1555|   847k|}
dav1d_recon_b_inter_8bpc:
 1559|  1.79M|{
 1560|  1.79M|    Dav1dTileState *const ts = t->ts;
 1561|  1.79M|    const Dav1dFrameContext *const f = t->f;
 1562|  1.79M|    const Dav1dDSPContext *const dsp = f->dsp;
 1563|  1.79M|    const int bx4 = t->bx & 31, by4 = t->by & 31;
 1564|  1.79M|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1565|  1.79M|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 1566|  1.79M|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
 1567|  1.79M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
 1568|  1.79M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
 1569|  1.79M|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
 1570|  1.79M|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (1570:28): [True: 773k, False: 1.01M]
  ------------------
 1571|   773k|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (1571:29): [True: 654k, False: 118k]
  |  Branch (1571:45): [True: 59.1k, False: 59.4k]
  ------------------
 1572|   713k|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (1572:29): [True: 625k, False: 88.4k]
  |  Branch (1572:45): [True: 44.1k, False: 44.2k]
  ------------------
 1573|  1.79M|    const int chr_layout_idx = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I400 ? 0 :
  ------------------
  |  Branch (1573:32): [True: 1.01M, False: 773k]
  ------------------
 1574|  1.79M|                               DAV1D_PIXEL_LAYOUT_I444 - f->cur.p.layout;
 1575|  1.79M|    int res;
 1576|       |
 1577|       |    // prediction
 1578|  1.79M|    const int cbh4 = (bh4 + ss_ver) >> ss_ver, cbw4 = (bw4 + ss_hor) >> ss_hor;
 1579|  1.79M|    pixel *dst = ((pixel *) f->cur.data[0]) +
 1580|  1.79M|        4 * (t->by * PXSTRIDE(f->cur.stride[0]) + t->bx);
  ------------------
  |  |   53|  1.79M|#define PXSTRIDE(x) (x)
  ------------------
 1581|  1.79M|    const ptrdiff_t uvdstoff =
 1582|  1.79M|        4 * ((t->bx >> ss_hor) + (t->by >> ss_ver) * PXSTRIDE(f->cur.stride[1]));
  ------------------
  |  |   53|  1.79M|#define PXSTRIDE(x) (x)
  ------------------
 1583|  1.79M|    if (IS_KEY_OR_INTRA(f->frame_hdr)) {
  ------------------
  |  |   43|  1.79M|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  1.79M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 14.2k, False: 1.77M]
  |  |  ------------------
  ------------------
 1584|       |        // intrabc
 1585|  14.2k|        assert(!f->frame_hdr->super_res.enabled);
  ------------------
  |  Branch (1585:9): [True: 14.2k, False: 0]
  ------------------
 1586|  14.2k|        res = mc(t, dst, NULL, f->cur.stride[0], bw4, bh4, t->bx, t->by, 0,
 1587|  14.2k|                 b->mv[0], &f->sr_cur, 0 /* unused */, FILTER_2D_BILINEAR);
 1588|  14.2k|        if (res) return res;
  ------------------
  |  Branch (1588:13): [True: 0, False: 14.2k]
  ------------------
 1589|  33.4k|        if (has_chroma) for (int pl = 1; pl < 3; pl++) {
  ------------------
  |  Branch (1589:13): [True: 11.1k, False: 3.13k]
  |  Branch (1589:42): [True: 22.3k, False: 11.1k]
  ------------------
 1590|  22.3k|            res = mc(t, ((pixel *)f->cur.data[pl]) + uvdstoff, NULL, f->cur.stride[1],
 1591|  22.3k|                     bw4 << (bw4 == ss_hor), bh4 << (bh4 == ss_ver),
 1592|  22.3k|                     t->bx & ~ss_hor, t->by & ~ss_ver, pl, b->mv[0],
 1593|  22.3k|                     &f->sr_cur, 0 /* unused */, FILTER_2D_BILINEAR);
 1594|  22.3k|            if (res) return res;
  ------------------
  |  Branch (1594:17): [True: 0, False: 22.3k]
  ------------------
 1595|  22.3k|        }
 1596|  1.77M|    } else if (b->comp_type == COMP_INTER_NONE) {
  ------------------
  |  Branch (1596:16): [True: 1.68M, False: 89.0k]
  ------------------
 1597|  1.68M|        const Dav1dThreadPicture *const refp = &f->refp[b->ref[0]];
 1598|  1.68M|        const enum Filter2d filter_2d = b->filter2d;
 1599|       |
 1600|  1.68M|        if (imin(bw4, bh4) > 1 &&
  ------------------
  |  Branch (1600:13): [True: 1.46M, False: 219k]
  ------------------
 1601|  1.46M|            ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1601:15): [True: 1.04M, False: 424k]
  |  Branch (1601:44): [True: 617k, False: 426k]
  ------------------
 1602|   852k|             (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1602:15): [True: 97.1k, False: 755k]
  |  Branch (1602:44): [True: 91.0k, False: 6.12k]
  ------------------
 1603|   708k|        {
 1604|   708k|            res = warp_affine(t, dst, NULL, f->cur.stride[0], b_dim, 0, refp,
 1605|   708k|                              b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1605:31): [True: 91.0k, False: 617k]
  ------------------
 1606|   708k|                                  &f->frame_hdr->gmv[b->ref[0]]);
 1607|   708k|            if (res) return res;
  ------------------
  |  Branch (1607:17): [True: 0, False: 708k]
  ------------------
 1608|   979k|        } else {
 1609|   979k|            res = mc(t, dst, NULL, f->cur.stride[0],
 1610|   979k|                     bw4, bh4, t->bx, t->by, 0, b->mv[0], refp, b->ref[0], filter_2d);
 1611|   979k|            if (res) return res;
  ------------------
  |  Branch (1611:17): [True: 0, False: 979k]
  ------------------
 1612|   979k|            if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1612:17): [True: 161k, False: 818k]
  ------------------
 1613|   161k|                res = obmc(t, dst, f->cur.stride[0], b_dim, 0, bx4, by4, w4, h4);
 1614|   161k|                if (res) return res;
  ------------------
  |  Branch (1614:21): [True: 0, False: 161k]
  ------------------
 1615|   161k|            }
 1616|   979k|        }
 1617|  1.68M|        if (b->interintra_type) {
  ------------------
  |  Branch (1617:13): [True: 60.4k, False: 1.62M]
  ------------------
 1618|  60.4k|            pixel *const tl_edge = bitfn(t->scratch.edge) + 32;
  ------------------
  |  |   51|  60.4k|#define bitfn(x) x##_8bpc
  ------------------
 1619|  60.4k|            enum IntraPredMode m = b->interintra_mode == II_SMOOTH_PRED ?
  ------------------
  |  Branch (1619:36): [True: 8.79k, False: 51.6k]
  ------------------
 1620|  51.6k|                                   SMOOTH_PRED : b->interintra_mode;
 1621|  60.4k|            pixel *const tmp = bitfn(t->scratch.interintra);
  ------------------
  |  |   51|  60.4k|#define bitfn(x) x##_8bpc
  ------------------
 1622|  60.4k|            int angle = 0;
 1623|  60.4k|            const pixel *top_sb_edge = NULL;
 1624|  60.4k|            if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1624:17): [True: 8.87k, False: 51.5k]
  ------------------
 1625|  8.87k|                top_sb_edge = f->ipred_edge[0];
 1626|  8.87k|                const int sby = t->by >> f->sb_shift;
 1627|  8.87k|                top_sb_edge += f->sb128w * 128 * (sby - 1);
 1628|  8.87k|            }
 1629|  60.4k|            m = bytefn(dav1d_prepare_intra_edges)(t->bx, t->bx > ts->tiling.col_start,
  ------------------
  |  |   87|  60.4k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  60.4k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1630|  60.4k|                                                  t->by, t->by > ts->tiling.row_start,
 1631|  60.4k|                                                  ts->tiling.col_end, ts->tiling.row_end,
 1632|  60.4k|                                                  0, dst, f->cur.stride[0], top_sb_edge,
 1633|  60.4k|                                                  m, &angle, bw4, bh4, 0, tl_edge
 1634|  60.4k|                                                  HIGHBD_CALL_SUFFIX);
 1635|  60.4k|            dsp->ipred.intra_pred[m](tmp, 4 * bw4 * sizeof(pixel),
 1636|  60.4k|                                     tl_edge, bw4 * 4, bh4 * 4, 0, 0, 0
 1637|  60.4k|                                     HIGHBD_CALL_SUFFIX);
 1638|  60.4k|            dsp->mc.blend(dst, f->cur.stride[0], tmp,
 1639|  60.4k|                          bw4 * 4, bh4 * 4, II_MASK(0, bs, b));
  ------------------
  |  |   83|  60.4k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   84|  60.4k|    (size_t)((b)->interintra_type == INTER_INTRA_BLEND ? \
  |  |  ------------------
  |  |  |  Branch (84:14): [True: 44.8k, False: 15.5k]
  |  |  ------------------
  |  |   85|  60.4k|    dav1d_masks.offsets[c][(bs)-BS_32x32].ii[(b)->interintra_mode] : \
  |  |   86|  60.4k|    dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[0][(b)->wedge_idx]) * 8))
  ------------------
 1640|  60.4k|        }
 1641|       |
 1642|  1.68M|        if (!has_chroma) goto skip_inter_chroma_pred;
  ------------------
  |  Branch (1642:13): [True: 1.11M, False: 576k]
  ------------------
 1643|       |
 1644|       |        // sub8x8 derivation
 1645|   576k|        int is_sub8x8 = bw4 == ss_hor || bh4 == ss_ver;
  ------------------
  |  Branch (1645:25): [True: 46.4k, False: 530k]
  |  Branch (1645:42): [True: 32.8k, False: 497k]
  ------------------
 1646|   576k|        refmvs_block *const *r;
 1647|   576k|        if (is_sub8x8) {
  ------------------
  |  Branch (1647:13): [True: 79.9k, False: 496k]
  ------------------
 1648|  79.9k|            assert(ss_hor == 1);
  ------------------
  |  Branch (1648:13): [True: 79.9k, False: 1]
  ------------------
 1649|  79.9k|            r = &t->rt.r[(t->by & 31) + 5];
 1650|  79.9k|            if (bw4 == 1) is_sub8x8 &= r[0][t->bx - 1].ref.ref[0] > 0;
  ------------------
  |  Branch (1650:17): [True: 47.0k, False: 32.8k]
  ------------------
 1651|  79.9k|            if (bh4 == ss_ver) is_sub8x8 &= r[-1][t->bx].ref.ref[0] > 0;
  ------------------
  |  Branch (1651:17): [True: 43.0k, False: 36.8k]
  ------------------
 1652|  79.9k|            if (bw4 == 1 && bh4 == ss_ver)
  ------------------
  |  Branch (1652:17): [True: 47.0k, False: 32.8k]
  |  Branch (1652:29): [True: 10.1k, False: 36.9k]
  ------------------
 1653|  10.1k|                is_sub8x8 &= r[-1][t->bx - 1].ref.ref[0] > 0;
 1654|  79.9k|        }
 1655|       |
 1656|       |        // chroma prediction
 1657|   576k|        if (is_sub8x8) {
  ------------------
  |  Branch (1657:13): [True: 71.6k, False: 504k]
  ------------------
 1658|  71.6k|            assert(ss_hor == 1);
  ------------------
  |  Branch (1658:13): [True: 71.6k, False: 0]
  ------------------
 1659|  71.6k|            ptrdiff_t h_off = 0, v_off = 0;
 1660|  71.6k|            if (bw4 == 1 && bh4 == ss_ver) {
  ------------------
  |  Branch (1660:17): [True: 42.1k, False: 29.4k]
  |  Branch (1660:29): [True: 8.46k, False: 33.6k]
  ------------------
 1661|  25.4k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1661:34): [True: 16.9k, False: 8.46k]
  ------------------
 1662|  16.9k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1663|  16.9k|                             NULL, f->cur.stride[1],
 1664|  16.9k|                             bw4, bh4, t->bx - 1, t->by - 1, 1 + pl,
 1665|  16.9k|                             r[-1][t->bx - 1].mv.mv[0],
 1666|  16.9k|                             &f->refp[r[-1][t->bx - 1].ref.ref[0] - 1],
 1667|  16.9k|                             r[-1][t->bx - 1].ref.ref[0] - 1,
 1668|  16.9k|                             t->frame_thread.pass != 2 ? t->tl_4x4_filter :
  ------------------
  |  Branch (1668:30): [True: 0, False: 16.9k]
  ------------------
 1669|  16.9k|                                 f->frame_thread.b[((t->by - 1) * f->b4_stride) + t->bx - 1].filter2d);
 1670|  16.9k|                    if (res) return res;
  ------------------
  |  Branch (1670:25): [True: 0, False: 16.9k]
  ------------------
 1671|  16.9k|                }
 1672|  8.46k|                v_off = 2 * PXSTRIDE(f->cur.stride[1]);
  ------------------
  |  |   53|  8.46k|#define PXSTRIDE(x) (x)
  ------------------
 1673|  8.46k|                h_off = 2;
 1674|  8.46k|            }
 1675|  71.6k|            if (bw4 == 1) {
  ------------------
  |  Branch (1675:17): [True: 42.1k, False: 29.4k]
  ------------------
 1676|  42.1k|                const enum Filter2d left_filter_2d =
 1677|  42.1k|                    dav1d_filter_2d[t->l.filter[1][by4]][t->l.filter[0][by4]];
 1678|   126k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1678:34): [True: 84.2k, False: 42.1k]
  ------------------
 1679|  84.2k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + v_off, NULL,
 1680|  84.2k|                             f->cur.stride[1], bw4, bh4, t->bx - 1,
 1681|  84.2k|                             t->by, 1 + pl, r[0][t->bx - 1].mv.mv[0],
 1682|  84.2k|                             &f->refp[r[0][t->bx - 1].ref.ref[0] - 1],
 1683|  84.2k|                             r[0][t->bx - 1].ref.ref[0] - 1,
 1684|  84.2k|                             t->frame_thread.pass != 2 ? left_filter_2d :
  ------------------
  |  Branch (1684:30): [True: 0, False: 84.2k]
  ------------------
 1685|  84.2k|                                 f->frame_thread.b[(t->by * f->b4_stride) + t->bx - 1].filter2d);
 1686|  84.2k|                    if (res) return res;
  ------------------
  |  Branch (1686:25): [True: 0, False: 84.2k]
  ------------------
 1687|  84.2k|                }
 1688|  42.1k|                h_off = 2;
 1689|  42.1k|            }
 1690|  71.6k|            if (bh4 == ss_ver) {
  ------------------
  |  Branch (1690:17): [True: 37.9k, False: 33.6k]
  ------------------
 1691|  37.9k|                const enum Filter2d top_filter_2d =
 1692|  37.9k|                    dav1d_filter_2d[t->a->filter[1][bx4]][t->a->filter[0][bx4]];
 1693|   113k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1693:34): [True: 75.9k, False: 37.9k]
  ------------------
 1694|  75.9k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + h_off, NULL,
 1695|  75.9k|                             f->cur.stride[1], bw4, bh4, t->bx, t->by - 1,
 1696|  75.9k|                             1 + pl, r[-1][t->bx].mv.mv[0],
 1697|  75.9k|                             &f->refp[r[-1][t->bx].ref.ref[0] - 1],
 1698|  75.9k|                             r[-1][t->bx].ref.ref[0] - 1,
 1699|  75.9k|                             t->frame_thread.pass != 2 ? top_filter_2d :
  ------------------
  |  Branch (1699:30): [True: 0, False: 75.9k]
  ------------------
 1700|  75.9k|                                 f->frame_thread.b[((t->by - 1) * f->b4_stride) + t->bx].filter2d);
 1701|  75.9k|                    if (res) return res;
  ------------------
  |  Branch (1701:25): [True: 0, False: 75.9k]
  ------------------
 1702|  75.9k|                }
 1703|  37.9k|                v_off = 2 * PXSTRIDE(f->cur.stride[1]);
  ------------------
  |  |   53|  37.9k|#define PXSTRIDE(x) (x)
  ------------------
 1704|  37.9k|            }
 1705|   214k|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1705:30): [True: 143k, False: 71.6k]
  ------------------
 1706|   143k|                res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + h_off + v_off, NULL, f->cur.stride[1],
 1707|   143k|                         bw4, bh4, t->bx, t->by, 1 + pl, b->mv[0],
 1708|   143k|                         refp, b->ref[0], filter_2d);
 1709|   143k|                if (res) return res;
  ------------------
  |  Branch (1709:21): [True: 0, False: 143k]
  ------------------
 1710|   143k|            }
 1711|   504k|        } else {
 1712|   504k|            if (imin(cbw4, cbh4) > 1 &&
  ------------------
  |  Branch (1712:17): [True: 254k, False: 250k]
  ------------------
 1713|   254k|                ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1713:19): [True: 34.3k, False: 219k]
  |  Branch (1713:48): [True: 11.8k, False: 22.4k]
  ------------------
 1714|   242k|                 (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1714:19): [True: 43.0k, False: 199k]
  |  Branch (1714:48): [True: 41.8k, False: 1.13k]
  ------------------
 1715|  53.7k|            {
 1716|   161k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1716:34): [True: 107k, False: 53.7k]
  ------------------
 1717|   107k|                    res = warp_affine(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff, NULL,
 1718|   107k|                                      f->cur.stride[1], b_dim, 1 + pl, refp,
 1719|   107k|                                      b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1719:39): [True: 83.7k, False: 23.6k]
  ------------------
 1720|   107k|                                          &f->frame_hdr->gmv[b->ref[0]]);
 1721|   107k|                    if (res) return res;
  ------------------
  |  Branch (1721:25): [True: 0, False: 107k]
  ------------------
 1722|   107k|                }
 1723|   451k|            } else {
 1724|  1.35M|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1724:34): [True: 902k, False: 451k]
  ------------------
 1725|   902k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1726|   902k|                             NULL, f->cur.stride[1],
 1727|   902k|                             bw4 << (bw4 == ss_hor), bh4 << (bh4 == ss_ver),
 1728|   902k|                             t->bx & ~ss_hor, t->by & ~ss_ver,
 1729|   902k|                             1 + pl, b->mv[0], refp, b->ref[0], filter_2d);
 1730|   902k|                    if (res) return res;
  ------------------
  |  Branch (1730:25): [True: 0, False: 902k]
  ------------------
 1731|   902k|                    if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1731:25): [True: 296k, False: 606k]
  ------------------
 1732|   296k|                        res = obmc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1733|   296k|                                   f->cur.stride[1], b_dim, 1 + pl, bx4, by4, w4, h4);
 1734|   296k|                        if (res) return res;
  ------------------
  |  Branch (1734:29): [True: 0, False: 296k]
  ------------------
 1735|   296k|                    }
 1736|   902k|                }
 1737|   451k|            }
 1738|   504k|            if (b->interintra_type) {
  ------------------
  |  Branch (1738:17): [True: 57.5k, False: 447k]
  ------------------
 1739|       |                // FIXME for 8x32 with 4:2:2 subsampling, this probably does
 1740|       |                // the wrong thing since it will select 4x16, not 4x32, as a
 1741|       |                // transform size...
 1742|  57.5k|                const uint8_t *const ii_mask = II_MASK(chr_layout_idx, bs, b);
  ------------------
  |  |   83|  57.5k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   84|  57.5k|    (size_t)((b)->interintra_type == INTER_INTRA_BLEND ? \
  |  |  ------------------
  |  |  |  Branch (84:14): [True: 42.6k, False: 14.8k]
  |  |  ------------------
  |  |   85|  57.5k|    dav1d_masks.offsets[c][(bs)-BS_32x32].ii[(b)->interintra_mode] : \
  |  |   86|  57.5k|    dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[0][(b)->wedge_idx]) * 8))
  ------------------
 1743|       |
 1744|   172k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1744:34): [True: 115k, False: 57.5k]
  ------------------
 1745|   115k|                    pixel *const tmp = bitfn(t->scratch.interintra);
  ------------------
  |  |   51|   115k|#define bitfn(x) x##_8bpc
  ------------------
 1746|   115k|                    pixel *const tl_edge = bitfn(t->scratch.edge) + 32;
  ------------------
  |  |   51|   115k|#define bitfn(x) x##_8bpc
  ------------------
 1747|   115k|                    enum IntraPredMode m =
 1748|   115k|                        b->interintra_mode == II_SMOOTH_PRED ?
  ------------------
  |  Branch (1748:25): [True: 16.4k, False: 98.6k]
  ------------------
 1749|  98.6k|                        SMOOTH_PRED : b->interintra_mode;
 1750|   115k|                    int angle = 0;
 1751|   115k|                    pixel *const uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff;
 1752|   115k|                    const pixel *top_sb_edge = NULL;
 1753|   115k|                    if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1753:25): [True: 17.2k, False: 97.8k]
  ------------------
 1754|  17.2k|                        top_sb_edge = f->ipred_edge[pl + 1];
 1755|  17.2k|                        const int sby = t->by >> f->sb_shift;
 1756|  17.2k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1757|  17.2k|                    }
 1758|   115k|                    m = bytefn(dav1d_prepare_intra_edges)(t->bx >> ss_hor,
  ------------------
  |  |   87|   115k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   115k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 1759|   115k|                                                          (t->bx >> ss_hor) >
 1760|   115k|                                                              (ts->tiling.col_start >> ss_hor),
 1761|   115k|                                                          t->by >> ss_ver,
 1762|   115k|                                                          (t->by >> ss_ver) >
 1763|   115k|                                                              (ts->tiling.row_start >> ss_ver),
 1764|   115k|                                                          ts->tiling.col_end >> ss_hor,
 1765|   115k|                                                          ts->tiling.row_end >> ss_ver,
 1766|   115k|                                                          0, uvdst, f->cur.stride[1],
 1767|   115k|                                                          top_sb_edge, m,
 1768|   115k|                                                          &angle, cbw4, cbh4, 0, tl_edge
 1769|   115k|                                                          HIGHBD_CALL_SUFFIX);
 1770|   115k|                    dsp->ipred.intra_pred[m](tmp, cbw4 * 4 * sizeof(pixel),
 1771|   115k|                                             tl_edge, cbw4 * 4, cbh4 * 4, 0, 0, 0
 1772|   115k|                                             HIGHBD_CALL_SUFFIX);
 1773|   115k|                    dsp->mc.blend(uvdst, f->cur.stride[1], tmp,
 1774|   115k|                                  cbw4 * 4, cbh4 * 4, ii_mask);
 1775|   115k|                }
 1776|  57.5k|            }
 1777|   504k|        }
 1778|       |
 1779|  1.68M|    skip_inter_chroma_pred: {}
 1780|  1.68M|        t->tl_4x4_filter = filter_2d;
 1781|  1.68M|    } else {
 1782|  89.0k|        const enum Filter2d filter_2d = b->filter2d;
 1783|       |        // Maximum super block size is 128x128
 1784|  89.0k|        int16_t (*tmp)[128 * 128] = t->scratch.compinter;
 1785|  89.0k|        int jnt_weight;
 1786|  89.0k|        uint8_t *const seg_mask = t->scratch.seg_mask;
 1787|  89.0k|        const uint8_t *mask;
 1788|       |
 1789|   268k|        for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1789:25): [True: 179k, False: 89.0k]
  ------------------
 1790|   179k|            const Dav1dThreadPicture *const refp = &f->refp[b->ref[i]];
 1791|       |
 1792|   179k|            if (b->inter_mode == GLOBALMV_GLOBALMV && f->gmv_warp_allowed[b->ref[i]]) {
  ------------------
  |  Branch (1792:17): [True: 5.66k, False: 173k]
  |  Branch (1792:55): [True: 2.42k, False: 3.24k]
  ------------------
 1793|  2.41k|                res = warp_affine(t, NULL, tmp[i], bw4 * 4, b_dim, 0, refp,
 1794|  2.41k|                                  &f->frame_hdr->gmv[b->ref[i]]);
 1795|  2.41k|                if (res) return res;
  ------------------
  |  Branch (1795:21): [True: 0, False: 2.41k]
  ------------------
 1796|   176k|            } else {
 1797|   176k|                res = mc(t, NULL, tmp[i], 0, bw4, bh4, t->bx, t->by, 0,
 1798|   176k|                         b->mv[i], refp, b->ref[i], filter_2d);
 1799|   176k|                if (res) return res;
  ------------------
  |  Branch (1799:21): [True: 0, False: 176k]
  ------------------
 1800|   176k|            }
 1801|   179k|        }
 1802|  89.0k|        switch (b->comp_type) {
  ------------------
  |  Branch (1802:17): [True: 89.7k, False: 18.4E]
  ------------------
 1803|  54.2k|        case COMP_INTER_AVG:
  ------------------
  |  Branch (1803:9): [True: 54.2k, False: 34.8k]
  ------------------
 1804|  54.2k|            dsp->mc.avg(dst, f->cur.stride[0], tmp[0], tmp[1],
 1805|  54.2k|                        bw4 * 4, bh4 * 4 HIGHBD_CALL_SUFFIX);
 1806|  54.2k|            break;
 1807|  12.5k|        case COMP_INTER_WEIGHTED_AVG:
  ------------------
  |  Branch (1807:9): [True: 12.5k, False: 76.4k]
  ------------------
 1808|  12.5k|            jnt_weight = f->jnt_weights[b->ref[0]][b->ref[1]];
 1809|  12.5k|            dsp->mc.w_avg(dst, f->cur.stride[0], tmp[0], tmp[1],
 1810|  12.5k|                          bw4 * 4, bh4 * 4, jnt_weight HIGHBD_CALL_SUFFIX);
 1811|  12.5k|            break;
 1812|  14.6k|        case COMP_INTER_SEG:
  ------------------
  |  Branch (1812:9): [True: 14.6k, False: 74.4k]
  ------------------
 1813|  14.6k|            dsp->mc.w_mask[chr_layout_idx](dst, f->cur.stride[0],
 1814|  14.6k|                                           tmp[b->mask_sign], tmp[!b->mask_sign],
 1815|  14.6k|                                           bw4 * 4, bh4 * 4, seg_mask,
 1816|  14.6k|                                           b->mask_sign HIGHBD_CALL_SUFFIX);
 1817|  14.6k|            mask = seg_mask;
 1818|  14.6k|            break;
 1819|  8.21k|        case COMP_INTER_WEDGE:
  ------------------
  |  Branch (1819:9): [True: 8.21k, False: 80.8k]
  ------------------
 1820|  8.21k|            mask = WEDGE_MASK(0, bs, 0, b->wedge_idx);
  ------------------
  |  |   89|  8.21k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   90|  8.21k|    (size_t)dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[sign][idx] * 8))
  ------------------
 1821|  8.21k|            dsp->mc.mask(dst, f->cur.stride[0],
 1822|  8.21k|                         tmp[b->mask_sign], tmp[!b->mask_sign],
 1823|  8.21k|                         bw4 * 4, bh4 * 4, mask HIGHBD_CALL_SUFFIX);
 1824|  8.21k|            if (has_chroma)
  ------------------
  |  Branch (1824:17): [True: 7.61k, False: 599]
  ------------------
 1825|  7.61k|                mask = WEDGE_MASK(chr_layout_idx, bs, b->mask_sign, b->wedge_idx);
  ------------------
  |  |   89|  7.61k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   90|  7.61k|    (size_t)dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[sign][idx] * 8))
  ------------------
 1826|  8.21k|            break;
 1827|  89.0k|        }
 1828|       |
 1829|       |        // chroma
 1830|   248k|        if (has_chroma) for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1830:13): [True: 82.9k, False: 6.56k]
  |  Branch (1830:42): [True: 165k, False: 82.9k]
  ------------------
 1831|   496k|            for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1831:29): [True: 330k, False: 165k]
  ------------------
 1832|   330k|                const Dav1dThreadPicture *const refp = &f->refp[b->ref[i]];
 1833|   330k|                if (b->inter_mode == GLOBALMV_GLOBALMV &&
  ------------------
  |  Branch (1833:21): [True: 8.34k, False: 322k]
  ------------------
 1834|  8.34k|                    imin(cbw4, cbh4) > 1 && f->gmv_warp_allowed[b->ref[i]])
  ------------------
  |  Branch (1834:21): [True: 4.15k, False: 4.19k]
  |  Branch (1834:45): [True: 1.73k, False: 2.41k]
  ------------------
 1835|  1.73k|                {
 1836|  1.73k|                    res = warp_affine(t, NULL, tmp[i], bw4 * 4 >> ss_hor,
 1837|  1.73k|                                      b_dim, 1 + pl,
 1838|  1.73k|                                      refp, &f->frame_hdr->gmv[b->ref[i]]);
 1839|  1.73k|                    if (res) return res;
  ------------------
  |  Branch (1839:25): [True: 0, False: 1.73k]
  ------------------
 1840|   329k|                } else {
 1841|   329k|                    res = mc(t, NULL, tmp[i], 0, bw4, bh4, t->bx, t->by,
 1842|   329k|                             1 + pl, b->mv[i], refp, b->ref[i], filter_2d);
 1843|   329k|                    if (res) return res;
  ------------------
  |  Branch (1843:25): [True: 0, False: 329k]
  ------------------
 1844|   329k|                }
 1845|   330k|            }
 1846|   165k|            pixel *const uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff;
 1847|   165k|            switch (b->comp_type) {
  ------------------
  |  Branch (1847:21): [True: 166k, False: 18.4E]
  ------------------
 1848|   101k|            case COMP_INTER_AVG:
  ------------------
  |  Branch (1848:13): [True: 101k, False: 63.9k]
  ------------------
 1849|   101k|                dsp->mc.avg(uvdst, f->cur.stride[1], tmp[0], tmp[1],
 1850|   101k|                            bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver
 1851|   101k|                            HIGHBD_CALL_SUFFIX);
 1852|   101k|                break;
 1853|  22.1k|            case COMP_INTER_WEIGHTED_AVG:
  ------------------
  |  Branch (1853:13): [True: 22.1k, False: 143k]
  ------------------
 1854|  22.1k|                dsp->mc.w_avg(uvdst, f->cur.stride[1], tmp[0], tmp[1],
 1855|  22.1k|                              bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver, jnt_weight
 1856|  22.1k|                              HIGHBD_CALL_SUFFIX);
 1857|  22.1k|                break;
 1858|  15.2k|            case COMP_INTER_WEDGE:
  ------------------
  |  Branch (1858:13): [True: 15.2k, False: 150k]
  ------------------
 1859|  42.3k|            case COMP_INTER_SEG:
  ------------------
  |  Branch (1859:13): [True: 27.1k, False: 138k]
  ------------------
 1860|  42.3k|                dsp->mc.mask(uvdst, f->cur.stride[1],
 1861|  42.3k|                             tmp[b->mask_sign], tmp[!b->mask_sign],
 1862|  42.3k|                             bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver, mask
 1863|  42.3k|                             HIGHBD_CALL_SUFFIX);
 1864|  42.3k|                break;
 1865|   165k|            }
 1866|   165k|        }
 1867|  89.4k|    }
 1868|       |
 1869|  1.79M|    if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  1.79M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.79M]
  |  |  ------------------
  |  |   35|  1.79M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.79M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                  if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1870|      0|        hex_dump(dst, f->cur.stride[0], b_dim[0] * 4, b_dim[1] * 4, "y-pred");
 1871|      0|        if (has_chroma) {
  ------------------
  |  Branch (1871:13): [True: 0, False: 0]
  ------------------
 1872|      0|            hex_dump(&((pixel *) f->cur.data[1])[uvdstoff], f->cur.stride[1],
 1873|      0|                     cbw4 * 4, cbh4 * 4, "u-pred");
 1874|      0|            hex_dump(&((pixel *) f->cur.data[2])[uvdstoff], f->cur.stride[1],
 1875|      0|                     cbw4 * 4, cbh4 * 4, "v-pred");
 1876|      0|        }
 1877|      0|    }
 1878|       |
 1879|  1.79M|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
 1880|       |
 1881|  1.79M|    if (b->skip) {
  ------------------
  |  Branch (1881:9): [True: 1.39M, False: 400k]
  ------------------
 1882|       |        // reset coef contexts
 1883|  1.39M|        BlockContext *const a = t->a;
 1884|  1.39M|        dav1d_memset_pow2[b_dim[2]](&a->lcoef[bx4], 0x40);
 1885|  1.39M|        dav1d_memset_pow2[b_dim[3]](&t->l.lcoef[by4], 0x40);
 1886|  1.39M|        if (has_chroma) {
  ------------------
  |  Branch (1886:13): [True: 343k, False: 1.04M]
  ------------------
 1887|   343k|            dav1d_memset_pow2_fn memset_cw = dav1d_memset_pow2[ulog2(cbw4)];
 1888|   343k|            dav1d_memset_pow2_fn memset_ch = dav1d_memset_pow2[ulog2(cbh4)];
 1889|   343k|            memset_cw(&a->ccoef[0][cbx4], 0x40);
 1890|   343k|            memset_cw(&a->ccoef[1][cbx4], 0x40);
 1891|   343k|            memset_ch(&t->l.ccoef[0][cby4], 0x40);
 1892|   343k|            memset_ch(&t->l.ccoef[1][cby4], 0x40);
 1893|   343k|        }
 1894|  1.39M|        return 0;
 1895|  1.39M|    }
 1896|       |
 1897|   400k|    const TxfmInfo *const uvtx = &dav1d_txfm_dimensions[b->uvtx];
 1898|   400k|    const TxfmInfo *const ytx = &dav1d_txfm_dimensions[b->max_ytx];
 1899|   400k|    const uint16_t tx_split[2] = { b->tx_split0, b->tx_split1 };
 1900|       |
 1901|   804k|    for (int init_y = 0; init_y < bh4; init_y += 16) {
  ------------------
  |  Branch (1901:26): [True: 403k, False: 400k]
  ------------------
 1902|   811k|        for (int init_x = 0; init_x < bw4; init_x += 16) {
  ------------------
  |  Branch (1902:30): [True: 407k, False: 403k]
  ------------------
 1903|       |            // coefficient coding & inverse transforms
 1904|   407k|            int y_off = !!init_y, y;
 1905|   407k|            dst += PXSTRIDE(f->cur.stride[0]) * 4 * init_y;
  ------------------
  |  |   53|   407k|#define PXSTRIDE(x) (x)
  ------------------
 1906|   820k|            for (y = init_y, t->by += init_y; y < imin(h4, init_y + 16);
  ------------------
  |  Branch (1906:47): [True: 412k, False: 407k]
  ------------------
 1907|   412k|                 y += ytx->h, y_off++)
 1908|   412k|            {
 1909|   412k|                int x, x_off = !!init_x;
 1910|   832k|                for (x = init_x, t->bx += init_x; x < imin(w4, init_x + 16);
  ------------------
  |  Branch (1910:51): [True: 419k, False: 412k]
  ------------------
 1911|   419k|                     x += ytx->w, x_off++)
 1912|   419k|                {
 1913|   419k|                    read_coef_tree(t, bs, b, b->max_ytx, 0, tx_split,
 1914|   419k|                                   x_off, y_off, &dst[x * 4]);
 1915|   419k|                    t->bx += ytx->w;
 1916|   419k|                }
 1917|   412k|                dst += PXSTRIDE(f->cur.stride[0]) * 4 * ytx->h;
  ------------------
  |  |   53|   412k|#define PXSTRIDE(x) (x)
  ------------------
 1918|   412k|                t->bx -= x;
 1919|   412k|                t->by += ytx->h;
 1920|   412k|            }
 1921|   407k|            dst -= PXSTRIDE(f->cur.stride[0]) * 4 * y;
  ------------------
  |  |   53|   407k|#define PXSTRIDE(x) (x)
  ------------------
 1922|   407k|            t->by -= y;
 1923|       |
 1924|       |            // chroma coefs and inverse transform
 1925|   995k|            if (has_chroma) for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1925:17): [True: 332k, False: 75.5k]
  |  Branch (1925:46): [True: 663k, False: 332k]
  ------------------
 1926|   663k|                pixel *uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff +
 1927|   663k|                    (PXSTRIDE(f->cur.stride[1]) * init_y * 4 >> ss_ver);
  ------------------
  |  |   53|   663k|#define PXSTRIDE(x) (x)
  ------------------
 1928|   663k|                for (y = init_y >> ss_ver, t->by += init_y;
 1929|  1.33M|                     y < imin(ch4, (init_y + 16) >> ss_ver); y += uvtx->h)
  ------------------
  |  Branch (1929:22): [True: 668k, False: 663k]
  ------------------
 1930|   668k|                {
 1931|   668k|                    int x;
 1932|   668k|                    for (x = init_x >> ss_hor, t->bx += init_x;
 1933|  1.33M|                         x < imin(cw4, (init_x + 16) >> ss_hor); x += uvtx->w)
  ------------------
  |  Branch (1933:26): [True: 670k, False: 668k]
  ------------------
 1934|   670k|                    {
 1935|   670k|                        coef *cf;
 1936|   670k|                        int eob;
 1937|   670k|                        enum TxfmType txtp;
 1938|   670k|                        if (t->frame_thread.pass) {
  ------------------
  |  Branch (1938:29): [True: 670k, False: 18.4E]
  ------------------
 1939|   670k|                            const int p = t->frame_thread.pass & 1;
 1940|   670k|                            const int cbi = *ts->frame_thread[p].cbi++;
 1941|   670k|                            cf = ts->frame_thread[p].cf;
 1942|   670k|                            ts->frame_thread[p].cf += uvtx->w * uvtx->h * 16;
 1943|   670k|                            eob  = cbi >> 5;
 1944|   670k|                            txtp = cbi & 0x1f;
 1945|  18.4E|                        } else {
 1946|  18.4E|                            uint8_t cf_ctx;
 1947|  18.4E|                            cf = bitfn(t->cf);
  ------------------
  |  |   51|  18.4E|#define bitfn(x) x##_8bpc
  ------------------
 1948|  18.4E|                            txtp = t->scratch.txtp_map[(by4 + (y << ss_ver)) * 32 +
 1949|  18.4E|                                                        bx4 + (x << ss_hor)];
 1950|  18.4E|                            eob = decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
 1951|  18.4E|                                               &t->l.ccoef[pl][cby4 + y],
 1952|  18.4E|                                               b->uvtx, bs, b, 0, 1 + pl,
 1953|  18.4E|                                               cf, &txtp, &cf_ctx);
 1954|  18.4E|                            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  18.4E|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 18.4E]
  |  |  ------------------
  |  |   35|  18.4E|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  18.4E|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1955|      0|                                printf("Post-uv-cf-blk[pl=%d,tx=%d,"
 1956|      0|                                       "txtp=%d,eob=%d]: r=%d\n",
 1957|      0|                                       pl, b->uvtx, txtp, eob, ts->msac.rng);
 1958|  18.4E|                            int ctw = imin(uvtx->w, (f->bw - t->bx + ss_hor) >> ss_hor);
 1959|  18.4E|                            int cth = imin(uvtx->h, (f->bh - t->by + ss_ver) >> ss_ver);
 1960|  18.4E|                            dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
 1961|  18.4E|                            dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
 1962|  18.4E|                        }
 1963|   670k|                        if (eob >= 0) {
  ------------------
  |  Branch (1963:29): [True: 200k, False: 470k]
  ------------------
 1964|   200k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   200k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 200k]
  |  |  ------------------
  |  |   35|   200k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   200k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1965|      0|                                coef_dump(cf, uvtx->h * 4, uvtx->w * 4, 3, "dq");
 1966|   200k|                            dsp->itx.itxfm_add[b->uvtx]
 1967|   200k|                                              [txtp](&uvdst[4 * x],
 1968|   200k|                                                     f->cur.stride[1],
 1969|   200k|                                                     cf, eob HIGHBD_CALL_SUFFIX);
 1970|   200k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   200k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 200k]
  |  |  ------------------
  |  |   35|   200k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   200k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1971|      0|                                hex_dump(&uvdst[4 * x], f->cur.stride[1],
 1972|      0|                                         uvtx->w * 4, uvtx->h * 4, "recon");
 1973|   200k|                        }
 1974|   670k|                        t->bx += uvtx->w << ss_hor;
 1975|   670k|                    }
 1976|   668k|                    uvdst += PXSTRIDE(f->cur.stride[1]) * 4 * uvtx->h;
  ------------------
  |  |   53|   668k|#define PXSTRIDE(x) (x)
  ------------------
 1977|   668k|                    t->bx -= x << ss_hor;
 1978|   668k|                    t->by += uvtx->h << ss_ver;
 1979|   668k|                }
 1980|   663k|                t->by -= y << ss_ver;
 1981|   663k|            }
 1982|   407k|        }
 1983|   403k|    }
 1984|   400k|    return 0;
 1985|  1.79M|}
dav1d_filter_sbrow_deblock_cols_8bpc:
 1987|   518k|void bytefn(dav1d_filter_sbrow_deblock_cols)(Dav1dFrameContext *const f, const int sby) {
 1988|   518k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_DEBLOCK) ||
  ------------------
  |  Branch (1988:9): [True: 3, False: 518k]
  ------------------
 1989|   518k|        (!f->frame_hdr->loopfilter.level_y[0] && !f->frame_hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (1989:10): [True: 5.04k, False: 513k]
  |  Branch (1989:50): [True: 0, False: 5.04k]
  ------------------
 1990|      0|    {
 1991|      0|        return;
 1992|      0|    }
 1993|   518k|    const int y = sby * f->sb_step * 4;
 1994|   518k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1995|   518k|    pixel *const p[3] = {
 1996|   518k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|   518k|#define PXSTRIDE(x) (x)
  ------------------
 1997|   518k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|   518k|#define PXSTRIDE(x) (x)
  ------------------
 1998|   518k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
  ------------------
  |  |   53|   518k|#define PXSTRIDE(x) (x)
  ------------------
 1999|   518k|    };
 2000|   518k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2001|   518k|    bytefn(dav1d_loopfilter_sbrow_cols)(f, p, mask, sby,
  ------------------
  |  |   87|   518k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   518k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2002|   518k|                                        f->lf.start_of_tile_row[sby]);
 2003|   518k|}
dav1d_filter_sbrow_deblock_rows_8bpc:
 2005|   566k|void bytefn(dav1d_filter_sbrow_deblock_rows)(Dav1dFrameContext *const f, const int sby) {
 2006|   566k|    const int y = sby * f->sb_step * 4;
 2007|   566k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2008|   566k|    pixel *const p[3] = {
 2009|   566k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|   566k|#define PXSTRIDE(x) (x)
  ------------------
 2010|   566k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|   566k|#define PXSTRIDE(x) (x)
  ------------------
 2011|   566k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
  ------------------
  |  |   53|   566k|#define PXSTRIDE(x) (x)
  ------------------
 2012|   566k|    };
 2013|   566k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2014|   566k|    if (f->c->inloop_filters & DAV1D_INLOOPFILTER_DEBLOCK &&
  ------------------
  |  Branch (2014:9): [True: 566k, False: 18.4E]
  ------------------
 2015|   566k|        (f->frame_hdr->loopfilter.level_y[0] || f->frame_hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (2015:10): [True: 513k, False: 52.8k]
  |  Branch (2015:49): [True: 5.02k, False: 47.8k]
  ------------------
 2016|   518k|    {
 2017|   518k|        bytefn(dav1d_loopfilter_sbrow_rows)(f, p, mask, sby);
  ------------------
  |  |   87|   518k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   518k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2018|   518k|    }
 2019|   566k|    if (f->seq_hdr->cdef || f->lf.restore_planes) {
  ------------------
  |  Branch (2019:9): [True: 72.3k, False: 494k]
  |  Branch (2019:29): [True: 168k, False: 325k]
  ------------------
 2020|       |        // Store loop filtered pixels required by CDEF / LR
 2021|   241k|        bytefn(dav1d_copy_lpf)(f, p, sby);
  ------------------
  |  |   87|   241k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   241k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2022|   241k|    }
 2023|   566k|}
dav1d_filter_sbrow_cdef_8bpc:
 2025|  72.3k|void bytefn(dav1d_filter_sbrow_cdef)(Dav1dTaskContext *const tc, const int sby) {
 2026|  72.3k|    const Dav1dFrameContext *const f = tc->f;
 2027|  72.3k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_CDEF)) return;
  ------------------
  |  Branch (2027:9): [True: 0, False: 72.3k]
  ------------------
 2028|  72.3k|    const int sbsz = f->sb_step;
 2029|  72.3k|    const int y = sby * sbsz * 4;
 2030|  72.3k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2031|  72.3k|    pixel *const p[3] = {
 2032|  72.3k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|  72.3k|#define PXSTRIDE(x) (x)
  ------------------
 2033|  72.3k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  72.3k|#define PXSTRIDE(x) (x)
  ------------------
 2034|  72.3k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
  ------------------
  |  |   53|  72.3k|#define PXSTRIDE(x) (x)
  ------------------
 2035|  72.3k|    };
 2036|  72.3k|    Av1Filter *prev_mask = f->lf.mask + ((sby - 1) >> !f->seq_hdr->sb128) * f->sb128w;
 2037|  72.3k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2038|  72.3k|    const int start = sby * sbsz;
 2039|  72.3k|    if (sby) {
  ------------------
  |  Branch (2039:9): [True: 57.5k, False: 14.7k]
  ------------------
 2040|  57.5k|        const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2041|  57.5k|        pixel *p_up[3] = {
 2042|  57.5k|            p[0] - 8 * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|  57.5k|#define PXSTRIDE(x) (x)
  ------------------
 2043|  57.5k|            p[1] - (8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  57.5k|#define PXSTRIDE(x) (x)
  ------------------
 2044|  57.5k|            p[2] - (8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  57.5k|#define PXSTRIDE(x) (x)
  ------------------
 2045|  57.5k|        };
 2046|  57.5k|        bytefn(dav1d_cdef_brow)(tc, p_up, prev_mask, start - 2, start, 1, sby);
  ------------------
  |  |   87|  57.5k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  57.5k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2047|  57.5k|    }
 2048|  72.3k|    const int n_blks = sbsz - 2 * (sby + 1 < f->sbh);
 2049|  72.3k|    const int end = imin(start + n_blks, f->bh);
 2050|  72.3k|    bytefn(dav1d_cdef_brow)(tc, p, mask, start, end, 0, sby);
  ------------------
  |  |   87|  72.3k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  72.3k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2051|  72.3k|}
dav1d_filter_sbrow_resize_8bpc:
 2053|  14.0k|void bytefn(dav1d_filter_sbrow_resize)(Dav1dFrameContext *const f, const int sby) {
 2054|  14.0k|    const int sbsz = f->sb_step;
 2055|  14.0k|    const int y = sby * sbsz * 4;
 2056|  14.0k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2057|  14.0k|    const pixel *const p[3] = {
 2058|  14.0k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
  ------------------
  |  |   53|  14.0k|#define PXSTRIDE(x) (x)
  ------------------
 2059|  14.0k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  14.0k|#define PXSTRIDE(x) (x)
  ------------------
 2060|  14.0k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
  ------------------
  |  |   53|  14.0k|#define PXSTRIDE(x) (x)
  ------------------
 2061|  14.0k|    };
 2062|  14.0k|    pixel *const sr_p[3] = {
 2063|  14.0k|        f->lf.sr_p[0] + y * PXSTRIDE(f->sr_cur.p.stride[0]),
  ------------------
  |  |   53|  14.0k|#define PXSTRIDE(x) (x)
  ------------------
 2064|  14.0k|        f->lf.sr_p[1] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver),
  ------------------
  |  |   53|  14.0k|#define PXSTRIDE(x) (x)
  ------------------
 2065|  14.0k|        f->lf.sr_p[2] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver)
  ------------------
  |  |   53|  14.0k|#define PXSTRIDE(x) (x)
  ------------------
 2066|  14.0k|    };
 2067|  14.0k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400;
 2068|  55.5k|    for (int pl = 0; pl < 1 + 2 * has_chroma; pl++) {
  ------------------
  |  Branch (2068:22): [True: 41.4k, False: 14.0k]
  ------------------
 2069|  41.4k|        const int ss_ver = pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (2069:28): [True: 27.3k, False: 14.0k]
  |  Branch (2069:34): [True: 13.8k, False: 13.4k]
  ------------------
 2070|  41.4k|        const int h_start = 8 * !!sby >> ss_ver;
 2071|  41.4k|        const ptrdiff_t dst_stride = f->sr_cur.p.stride[!!pl];
 2072|  41.4k|        pixel *dst = sr_p[pl] - h_start * PXSTRIDE(dst_stride);
  ------------------
  |  |   53|  41.4k|#define PXSTRIDE(x) (x)
  ------------------
 2073|  41.4k|        const ptrdiff_t src_stride = f->cur.stride[!!pl];
 2074|  41.4k|        const pixel *src = p[pl] - h_start * PXSTRIDE(src_stride);
  ------------------
  |  |   53|  41.4k|#define PXSTRIDE(x) (x)
  ------------------
 2075|  41.4k|        const int h_end = 4 * (sbsz - 2 * (sby + 1 < f->sbh)) >> ss_ver;
 2076|  41.4k|        const int ss_hor = pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (2076:28): [True: 27.3k, False: 14.0k]
  |  Branch (2076:34): [True: 24.3k, False: 2.98k]
  ------------------
 2077|  41.4k|        const int dst_w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
 2078|  41.4k|        const int src_w = (4 * f->bw + ss_hor) >> ss_hor;
 2079|  41.4k|        const int img_h = (f->cur.p.h - sbsz * 4 * sby + ss_ver) >> ss_ver;
 2080|       |
 2081|  41.4k|        f->dsp->mc.resize(dst, dst_stride, src, src_stride, dst_w,
 2082|  41.4k|                          imin(img_h, h_end) + h_start, src_w,
 2083|  41.4k|                          f->resize_step[!!pl], f->resize_start[!!pl]
 2084|  41.4k|                          HIGHBD_CALL_SUFFIX);
 2085|  41.4k|    }
 2086|  14.0k|}
dav1d_filter_sbrow_lr_8bpc:
 2088|   175k|void bytefn(dav1d_filter_sbrow_lr)(Dav1dFrameContext *const f, const int sby) {
 2089|   175k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_RESTORATION)) return;
  ------------------
  |  Branch (2089:9): [True: 0, False: 175k]
  ------------------
 2090|   175k|    const int y = sby * f->sb_step * 4;
 2091|   175k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2092|   175k|    pixel *const sr_p[3] = {
 2093|   175k|        f->lf.sr_p[0] + y * PXSTRIDE(f->sr_cur.p.stride[0]),
  ------------------
  |  |   53|   175k|#define PXSTRIDE(x) (x)
  ------------------
 2094|   175k|        f->lf.sr_p[1] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver),
  ------------------
  |  |   53|   175k|#define PXSTRIDE(x) (x)
  ------------------
 2095|   175k|        f->lf.sr_p[2] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver)
  ------------------
  |  |   53|   175k|#define PXSTRIDE(x) (x)
  ------------------
 2096|   175k|    };
 2097|   175k|    bytefn(dav1d_lr_sbrow)(f, sr_p, sby);
  ------------------
  |  |   87|   175k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   175k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2098|   175k|}
dav1d_backup_ipred_edge_8bpc:
 2111|   984k|void bytefn(dav1d_backup_ipred_edge)(Dav1dTaskContext *const t) {
 2112|   984k|    const Dav1dFrameContext *const f = t->f;
 2113|   984k|    Dav1dTileState *const ts = t->ts;
 2114|   984k|    const int sby = t->by >> f->sb_shift;
 2115|   984k|    const int sby_off = f->sb128w * 128 * sby;
 2116|   984k|    const int x_off = ts->tiling.col_start;
 2117|       |
 2118|   984k|    const pixel *const y =
 2119|   984k|        ((const pixel *) f->cur.data[0]) + x_off * 4 +
 2120|   984k|                    ((t->by + f->sb_step) * 4 - 1) * PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|   984k|#define PXSTRIDE(x) (x)
  ------------------
 2121|   984k|    pixel_copy(&f->ipred_edge[0][sby_off + x_off * 4], y,
  ------------------
  |  |   47|   984k|#define pixel_copy memcpy
  ------------------
 2122|   984k|               4 * (ts->tiling.col_end - x_off));
 2123|       |
 2124|   984k|    if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (2124:9): [True: 101k, False: 883k]
  ------------------
 2125|   101k|        const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2126|   101k|        const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 2127|       |
 2128|   101k|        const ptrdiff_t uv_off = (x_off * 4 >> ss_hor) +
 2129|   101k|            (((t->by + f->sb_step) * 4 >> ss_ver) - 1) * PXSTRIDE(f->cur.stride[1]);
  ------------------
  |  |   53|   101k|#define PXSTRIDE(x) (x)
  ------------------
 2130|   302k|        for (int pl = 1; pl <= 2; pl++)
  ------------------
  |  Branch (2130:26): [True: 201k, False: 101k]
  ------------------
 2131|   201k|            pixel_copy(&f->ipred_edge[pl][sby_off + (x_off * 4 >> ss_hor)],
  ------------------
  |  |   47|   201k|#define pixel_copy memcpy
  ------------------
 2132|   201k|                       &((const pixel *) f->cur.data[pl])[uv_off],
 2133|   201k|                       4 * (ts->tiling.col_end - x_off) >> ss_hor);
 2134|   101k|    }
 2135|   984k|}
dav1d_copy_pal_block_y_8bpc:
 2141|   112k|{
 2142|   112k|    const Dav1dFrameContext *const f = t->f;
 2143|   112k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2143:24): [True: 112k, False: 1]
  ------------------
 2144|   112k|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2145|   112k|                            ((t->bx >> 1) + (t->by & 1))][0] :
 2146|   112k|        bytefn(t->scratch.pal)[0];
  ------------------
  |  |   87|      1|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   112k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2147|   583k|    for (int x = 0; x < bw4; x++)
  ------------------
  |  Branch (2147:21): [True: 470k, False: 112k]
  ------------------
 2148|   470k|        memcpy(bytefn(t->al_pal)[0][bx4 + x][0], pal, 8 * sizeof(pixel));
  ------------------
  |  |   87|   470k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   470k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2149|   483k|    for (int y = 0; y < bh4; y++)
  ------------------
  |  Branch (2149:21): [True: 370k, False: 112k]
  ------------------
 2150|   370k|        memcpy(bytefn(t->al_pal)[1][by4 + y][0], pal, 8 * sizeof(pixel));
  ------------------
  |  |   87|   370k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   370k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2151|   112k|}
dav1d_copy_pal_block_uv_8bpc:
 2157|  20.1k|{
 2158|  20.1k|    const Dav1dFrameContext *const f = t->f;
 2159|  20.1k|    const pixel (*const pal)[8] = t->frame_thread.pass ?
  ------------------
  |  Branch (2159:35): [True: 20.1k, False: 0]
  ------------------
 2160|  20.1k|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2161|  20.1k|                            ((t->bx >> 1) + (t->by & 1))] :
 2162|  20.1k|        bytefn(t->scratch.pal);
  ------------------
  |  |   87|      0|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|      0|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2163|       |    // see aomedia bug 2183 for why we use luma coordinates here
 2164|  60.3k|    for (int pl = 1; pl <= 2; pl++) {
  ------------------
  |  Branch (2164:22): [True: 40.2k, False: 20.1k]
  ------------------
 2165|   260k|        for (int x = 0; x < bw4; x++)
  ------------------
  |  Branch (2165:25): [True: 220k, False: 40.2k]
  ------------------
 2166|   220k|            memcpy(bytefn(t->al_pal)[0][bx4 + x][pl], pal[pl], 8 * sizeof(pixel));
  ------------------
  |  |   87|   220k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   220k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2167|   260k|        for (int y = 0; y < bh4; y++)
  ------------------
  |  Branch (2167:25): [True: 220k, False: 40.2k]
  ------------------
 2168|   220k|            memcpy(bytefn(t->al_pal)[1][by4 + y][pl], pal[pl], 8 * sizeof(pixel));
  ------------------
  |  |   87|   220k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   220k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2169|  40.2k|    }
 2170|  20.1k|}
dav1d_read_pal_plane_8bpc:
 2175|   132k|{
 2176|   132k|    Dav1dTileState *const ts = t->ts;
 2177|   132k|    const Dav1dFrameContext *const f = t->f;
 2178|   132k|    const int pal_sz = b->pal_sz[pl] = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|   132k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 2179|   132k|                                           ts->cdf.m.pal_sz[pl][sz_ctx], 6) + 2;
 2180|   132k|    pixel cache[16], used_cache[8];
 2181|   132k|    int l_cache = pl ? t->pal_sz_uv[1][by4] : t->l.pal_sz[by4];
  ------------------
  |  Branch (2181:19): [True: 20.1k, False: 112k]
  ------------------
 2182|   132k|    int n_cache = 0;
 2183|       |    // don't reuse above palette outside SB64 boundaries
 2184|   132k|    int a_cache = by4 & 15 ? pl ? t->pal_sz_uv[0][bx4] : t->a->pal_sz[bx4] : 0;
  ------------------
  |  Branch (2184:19): [True: 105k, False: 26.6k]
  |  Branch (2184:30): [True: 11.8k, False: 93.9k]
  ------------------
 2185|   132k|    const pixel *l = bytefn(t->al_pal)[1][by4][pl];
  ------------------
  |  |   87|   132k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   132k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2186|   132k|    const pixel *a = bytefn(t->al_pal)[0][bx4][pl];
  ------------------
  |  |   87|   132k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   132k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2187|       |
 2188|       |    // fill/sort cache
 2189|   260k|    while (l_cache && a_cache) {
  ------------------
  |  Branch (2189:12): [True: 168k, False: 91.6k]
  |  Branch (2189:23): [True: 128k, False: 40.8k]
  ------------------
 2190|   128k|        if (*l < *a) {
  ------------------
  |  Branch (2190:13): [True: 36.4k, False: 91.5k]
  ------------------
 2191|  36.4k|            if (!n_cache || cache[n_cache - 1] != *l)
  ------------------
  |  Branch (2191:17): [True: 7.46k, False: 29.0k]
  |  Branch (2191:29): [True: 28.9k, False: 99]
  ------------------
 2192|  36.3k|                cache[n_cache++] = *l;
 2193|  36.4k|            l++;
 2194|  36.4k|            l_cache--;
 2195|  91.5k|        } else {
 2196|  91.5k|            if (*a == *l) {
  ------------------
  |  Branch (2196:17): [True: 47.9k, False: 43.5k]
  ------------------
 2197|  47.9k|                l++;
 2198|  47.9k|                l_cache--;
 2199|  47.9k|            }
 2200|  91.5k|            if (!n_cache || cache[n_cache - 1] != *a)
  ------------------
  |  Branch (2200:17): [True: 14.6k, False: 76.8k]
  |  Branch (2200:29): [True: 71.8k, False: 5.03k]
  ------------------
 2201|  86.4k|                cache[n_cache++] = *a;
 2202|  91.5k|            a++;
 2203|  91.5k|            a_cache--;
 2204|  91.5k|        }
 2205|   128k|    }
 2206|   132k|    if (l_cache) {
  ------------------
  |  Branch (2206:9): [True: 40.8k, False: 91.6k]
  ------------------
 2207|   162k|        do {
 2208|   162k|            if (!n_cache || cache[n_cache - 1] != *l)
  ------------------
  |  Branch (2208:17): [True: 33.3k, False: 128k]
  |  Branch (2208:29): [True: 109k, False: 18.9k]
  ------------------
 2209|   143k|                cache[n_cache++] = *l;
 2210|   162k|            l++;
 2211|   162k|        } while (--l_cache > 0);
  ------------------
  |  Branch (2211:18): [True: 121k, False: 40.8k]
  ------------------
 2212|  91.6k|    } else if (a_cache) {
  ------------------
  |  Branch (2212:16): [True: 41.3k, False: 50.2k]
  ------------------
 2213|   175k|        do {
 2214|   175k|            if (!n_cache || cache[n_cache - 1] != *a)
  ------------------
  |  Branch (2214:17): [True: 27.8k, False: 147k]
  |  Branch (2214:29): [True: 91.7k, False: 55.4k]
  ------------------
 2215|   119k|                cache[n_cache++] = *a;
 2216|   175k|            a++;
 2217|   175k|        } while (--a_cache > 0);
  ------------------
  |  Branch (2217:18): [True: 133k, False: 41.3k]
  ------------------
 2218|  41.3k|    }
 2219|       |
 2220|       |    // find reused cache entries
 2221|   132k|    int i = 0;
 2222|   453k|    for (int n = 0; n < n_cache && i < pal_sz; n++)
  ------------------
  |  Branch (2222:21): [True: 338k, False: 115k]
  |  Branch (2222:36): [True: 320k, False: 17.4k]
  ------------------
 2223|   320k|        if (dav1d_msac_decode_bool_equi(&ts->msac))
  ------------------
  |  |   53|   320k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2223:13): [True: 174k, False: 146k]
  ------------------
 2224|   174k|            used_cache[i++] = cache[n];
 2225|   132k|    const int n_used_cache = i;
 2226|       |
 2227|       |    // parse new entries
 2228|   132k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2228:24): [True: 132k, False: 8]
  ------------------
 2229|   132k|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2230|   132k|                            ((t->bx >> 1) + (t->by & 1))][pl] :
 2231|   132k|        bytefn(t->scratch.pal)[pl];
  ------------------
  |  |   87|      8|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|   132k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2232|   132k|    if (i < pal_sz) {
  ------------------
  |  Branch (2232:9): [True: 112k, False: 19.7k]
  ------------------
 2233|   112k|        const int bpc = BITDEPTH == 8 ? 8 : f->cur.p.bpc;
  ------------------
  |  Branch (2233:25): [True: 112k, Folded]
  ------------------
 2234|   112k|        int prev = pal[i++] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2235|       |
 2236|   112k|        if (i < pal_sz) {
  ------------------
  |  Branch (2236:13): [True: 94.4k, False: 18.3k]
  ------------------
 2237|  94.4k|            int bits = bpc - 3 + dav1d_msac_decode_bools(&ts->msac, 2);
 2238|  94.4k|            const int max = (1 << bpc) - 1;
 2239|       |
 2240|   206k|            do {
 2241|   206k|                const int delta = dav1d_msac_decode_bools(&ts->msac, bits);
 2242|   206k|                prev = pal[i++] = imin(prev + delta + !pl, max);
 2243|   206k|                if (prev + !pl >= max) {
  ------------------
  |  Branch (2243:21): [True: 56.8k, False: 149k]
  ------------------
 2244|   146k|                    for (; i < pal_sz; i++)
  ------------------
  |  Branch (2244:28): [True: 90.0k, False: 56.8k]
  ------------------
 2245|  90.0k|                        pal[i] = max;
 2246|  56.8k|                    break;
 2247|  56.8k|                }
 2248|   149k|                bits = imin(bits, 1 + ulog2(max - prev - !pl));
 2249|   149k|            } while (i < pal_sz);
  ------------------
  |  Branch (2249:22): [True: 111k, False: 37.6k]
  ------------------
 2250|  94.4k|        }
 2251|       |
 2252|       |        // merge cache+new entries
 2253|   112k|        int n = 0, m = n_used_cache;
 2254|   642k|        for (i = 0; i < pal_sz; i++) {
  ------------------
  |  Branch (2254:21): [True: 529k, False: 112k]
  ------------------
 2255|   529k|            if (n < n_used_cache && (m >= pal_sz || used_cache[n] <= pal[m])) {
  ------------------
  |  Branch (2255:17): [True: 214k, False: 315k]
  |  Branch (2255:38): [True: 42.9k, False: 171k]
  |  Branch (2255:53): [True: 77.6k, False: 94.1k]
  ------------------
 2256|   120k|                pal[i] = used_cache[n++];
 2257|   409k|            } else {
 2258|   409k|                assert(m < pal_sz);
  ------------------
  |  Branch (2258:17): [True: 409k, False: 18.4E]
  ------------------
 2259|   409k|                pal[i] = pal[m++];
 2260|   409k|            }
 2261|   529k|        }
 2262|   112k|    } else {
 2263|  19.7k|        memcpy(pal, used_cache, n_used_cache * sizeof(*used_cache));
 2264|  19.7k|    }
 2265|       |
 2266|   132k|    if (DEBUG_BLOCK_INFO) {
  ------------------
  |  |   34|   132k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 132k]
  |  |  ------------------
  |  |   35|   132k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   132k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2267|      0|        printf("Post-pal[pl=%d,sz=%d,cache_size=%d,used_cache=%d]: r=%d, cache=",
 2268|      0|               pl, pal_sz, n_cache, n_used_cache, ts->msac.rng);
 2269|      0|        for (int n = 0; n < n_cache; n++)
  ------------------
  |  Branch (2269:25): [True: 0, False: 0]
  ------------------
 2270|      0|            printf("%c%02x", n ? ' ' : '[', cache[n]);
  ------------------
  |  Branch (2270:30): [True: 0, False: 0]
  ------------------
 2271|      0|        printf("%s, pal=", n_cache ? "]" : "[]");
  ------------------
  |  Branch (2271:28): [True: 0, False: 0]
  ------------------
 2272|      0|        for (int n = 0; n < pal_sz; n++)
  ------------------
  |  Branch (2272:25): [True: 0, False: 0]
  ------------------
 2273|      0|            printf("%c%02x", n ? ' ' : '[', pal[n]);
  ------------------
  |  Branch (2273:30): [True: 0, False: 0]
  ------------------
 2274|      0|        printf("]\n");
 2275|      0|    }
 2276|   132k|}
dav1d_read_pal_uv_8bpc:
 2280|  20.1k|{
 2281|  20.1k|    bytefn(dav1d_read_pal_plane)(t, b, 1, sz_ctx, bx4, by4);
  ------------------
  |  |   87|  20.1k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  20.1k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2282|       |
 2283|       |    // V pal coding
 2284|  20.1k|    Dav1dTileState *const ts = t->ts;
 2285|  20.1k|    const Dav1dFrameContext *const f = t->f;
 2286|  20.1k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2286:24): [True: 20.1k, False: 1]
  ------------------
 2287|  20.1k|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2288|  20.1k|                            ((t->bx >> 1) + (t->by & 1))][2] :
 2289|  20.1k|        bytefn(t->scratch.pal)[2];
  ------------------
  |  |   87|      1|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   51|  20.1k|#define bitfn(x) x##_8bpc
  |  |  ------------------
  ------------------
 2290|  20.1k|    const int bpc = BITDEPTH == 8 ? 8 : f->cur.p.bpc;
  ------------------
  |  Branch (2290:21): [True: 20.1k, Folded]
  ------------------
 2291|  20.1k|    if (dav1d_msac_decode_bool_equi(&ts->msac)) {
  ------------------
  |  |   53|  20.1k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2291:9): [True: 11.2k, False: 8.87k]
  ------------------
 2292|  11.2k|        const int bits = bpc - 4 + dav1d_msac_decode_bools(&ts->msac, 2);
 2293|  11.2k|        int prev = pal[0] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2294|  11.2k|        const int max = (1 << bpc) - 1;
 2295|  45.2k|        for (int i = 1; i < b->pal_sz[1]; i++) {
  ------------------
  |  Branch (2295:25): [True: 33.9k, False: 11.2k]
  ------------------
 2296|  33.9k|            int delta = dav1d_msac_decode_bools(&ts->msac, bits);
 2297|  33.9k|            if (delta && dav1d_msac_decode_bool_equi(&ts->msac)) delta = -delta;
  ------------------
  |  |   53|  33.5k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2297:17): [True: 33.5k, False: 423]
  |  Branch (2297:26): [True: 18.6k, False: 14.8k]
  ------------------
 2298|  33.9k|            prev = pal[i] = (prev + delta) & max;
 2299|  33.9k|        }
 2300|  11.2k|    } else {
 2301|  49.2k|        for (int i = 0; i < b->pal_sz[1]; i++)
  ------------------
  |  Branch (2301:25): [True: 40.3k, False: 8.87k]
  ------------------
 2302|  40.3k|            pal[i] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2303|  8.87k|    }
 2304|  20.1k|    if (DEBUG_BLOCK_INFO) {
  ------------------
  |  |   34|  20.1k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 20.1k]
  |  |  ------------------
  |  |   35|  20.1k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  20.1k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2305|      0|        printf("Post-pal[pl=2]: r=%d ", ts->msac.rng);
 2306|      0|        for (int n = 0; n < b->pal_sz[1]; n++)
  ------------------
  |  Branch (2306:25): [True: 0, False: 0]
  ------------------
 2307|      0|            printf("%c%02x", n ? ' ' : '[', pal[n]);
  ------------------
  |  Branch (2307:30): [True: 0, False: 0]
  ------------------
 2308|      0|        printf("]\n");
 2309|      0|    }
 2310|  20.1k|}
recon_tmpl.c:read_coef_tree:
  736|  7.57M|{
  737|  7.57M|    const Dav1dFrameContext *const f = t->f;
  738|  7.57M|    Dav1dTileState *const ts = t->ts;
  739|  7.57M|    const Dav1dDSPContext *const dsp = f->dsp;
  740|  7.57M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[ytx];
  741|  7.57M|    const int txw = t_dim->w, txh = t_dim->h;
  742|       |
  743|       |    /* y_off can be larger than 3 since lossless blocks use TX_4X4 but can't
  744|       |     * be splitted. Aviods an undefined left shift. */
  745|  7.57M|    if (depth < 2 && tx_split[depth] &&
  ------------------
  |  Branch (745:9): [True: 7.24M, False: 327k]
  |  Branch (745:22): [True: 428k, False: 6.81M]
  ------------------
  746|   428k|        tx_split[depth] & (1 << (y_off * 4 + x_off)))
  ------------------
  |  Branch (746:9): [True: 350k, False: 78.6k]
  ------------------
  747|   350k|    {
  748|   350k|        const enum RectTxfmSize sub = t_dim->sub;
  749|   350k|        const TxfmInfo *const sub_t_dim = &dav1d_txfm_dimensions[sub];
  750|   350k|        const int txsw = sub_t_dim->w, txsh = sub_t_dim->h;
  751|       |
  752|   350k|        read_coef_tree(t, bs, b, sub, depth + 1, tx_split,
  753|   350k|                       x_off * 2 + 0, y_off * 2 + 0, dst);
  754|   350k|        t->bx += txsw;
  755|   350k|        if (txw >= txh && t->bx < f->bw)
  ------------------
  |  Branch (755:13): [True: 269k, False: 80.3k]
  |  Branch (755:27): [True: 268k, False: 1.02k]
  ------------------
  756|   268k|            read_coef_tree(t, bs, b, sub, depth + 1, tx_split, x_off * 2 + 1,
  757|   268k|                           y_off * 2 + 0, dst ? &dst[4 * txsw] : NULL);
  ------------------
  |  Branch (757:43): [True: 100k, False: 168k]
  ------------------
  758|   350k|        t->bx -= txsw;
  759|   350k|        t->by += txsh;
  760|   350k|        if (txh >= txw && t->by < f->bh) {
  ------------------
  |  Branch (760:13): [True: 256k, False: 93.7k]
  |  Branch (760:27): [True: 254k, False: 1.53k]
  ------------------
  761|   254k|            if (dst)
  ------------------
  |  Branch (761:17): [True: 96.7k, False: 158k]
  ------------------
  762|  96.7k|                dst += 4 * txsh * PXSTRIDE(f->cur.stride[0]);
  ------------------
  |  |   53|  96.7k|#define PXSTRIDE(x) (x)
  ------------------
  763|   254k|            read_coef_tree(t, bs, b, sub, depth + 1, tx_split,
  764|   254k|                           x_off * 2 + 0, y_off * 2 + 1, dst);
  765|   254k|            t->bx += txsw;
  766|   254k|            if (txw >= txh && t->bx < f->bw)
  ------------------
  |  Branch (766:17): [True: 175k, False: 79.8k]
  |  Branch (766:31): [True: 174k, False: 999]
  ------------------
  767|   174k|                read_coef_tree(t, bs, b, sub, depth + 1, tx_split, x_off * 2 + 1,
  768|   174k|                               y_off * 2 + 1, dst ? &dst[4 * txsw] : NULL);
  ------------------
  |  Branch (768:47): [True: 62.9k, False: 111k]
  ------------------
  769|   254k|            t->bx -= txsw;
  770|   254k|        }
  771|   350k|        t->by -= txsh;
  772|  7.22M|    } else {
  773|  7.22M|        const int bx4 = t->bx & 31, by4 = t->by & 31;
  774|  7.22M|        enum TxfmType txtp;
  775|  7.22M|        uint8_t cf_ctx;
  776|  7.22M|        int eob;
  777|  7.22M|        coef *cf;
  778|       |
  779|  7.22M|        if (t->frame_thread.pass) {
  ------------------
  |  Branch (779:13): [True: 7.22M, False: 18.4E]
  ------------------
  780|  7.22M|            const int p = t->frame_thread.pass & 1;
  781|  7.22M|            assert(ts->frame_thread[p].cf);
  ------------------
  |  Branch (781:13): [True: 7.22M, False: 18.4E]
  ------------------
  782|  7.22M|            cf = ts->frame_thread[p].cf;
  783|  7.22M|            ts->frame_thread[p].cf += imin(t_dim->w, 8) * imin(t_dim->h, 8) * 16;
  784|  18.4E|        } else {
  785|  18.4E|            cf = bitfn(t->cf);
  ------------------
  |  |   51|  18.4E|#define bitfn(x) x##_8bpc
  ------------------
  786|  18.4E|        }
  787|  7.22M|        if (t->frame_thread.pass != 2) {
  ------------------
  |  Branch (787:13): [True: 6.36M, False: 859k]
  ------------------
  788|  6.36M|            eob = decode_coefs(t, &t->a->lcoef[bx4], &t->l.lcoef[by4],
  789|  6.36M|                               ytx, bs, b, 0, 0, cf, &txtp, &cf_ctx);
  790|  6.36M|            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  6.36M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 6.36M]
  |  |  ------------------
  |  |   35|  6.36M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  6.36M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  791|      0|                printf("Post-y-cf-blk[tx=%d,txtp=%d,eob=%d]: r=%d\n",
  792|      0|                       ytx, txtp, eob, ts->msac.rng);
  793|  6.36M|            dav1d_memset_likely_pow2(&t->a->lcoef[bx4], cf_ctx, imin(txw, f->bw - t->bx));
  794|  6.36M|            dav1d_memset_likely_pow2(&t->l.lcoef[by4], cf_ctx, imin(txh, f->bh - t->by));
  795|  6.36M|#define set_ctx(rep_macro) \
  796|  6.36M|            for (int y = 0; y < txh; y++) { \
  797|  6.36M|                rep_macro(txtp_map, 0, txtp); \
  798|  6.36M|                txtp_map += 32; \
  799|  6.36M|            }
  800|  6.36M|            uint8_t *txtp_map = &t->scratch.txtp_map[by4 * 32 + bx4];
  801|  6.36M|            case_set_upto16(t_dim->lw);
  ------------------
  |  |   80|  6.36M|    switch (var) { \
  |  |   81|  4.49M|    case 0: set_ctx(set_ctx1); break; \
  |  |  ------------------
  |  |  |  |  796|  9.29M|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 4.79M, False: 4.49M]
  |  |  |  |  ------------------
  |  |  |  |  797|  4.79M|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|  4.79M|    case 0: set_ctx(set_ctx1); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   56|  4.79M|    ((union alias8 *) &(var)[off])->u8 = (val) * 0x01
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|  4.79M|                txtp_map += 32; \
  |  |  |  |  799|  4.79M|            }
  |  |  ------------------
  |  |  |  Branch (81:5): [True: 4.49M, False: 1.86M]
  |  |  ------------------
  |  |   82|   790k|    case 1: set_ctx(set_ctx2); break; \
  |  |  ------------------
  |  |  |  |  796|  2.76M|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 1.97M, False: 790k]
  |  |  |  |  ------------------
  |  |  |  |  797|  1.97M|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   82|  1.97M|    case 1: set_ctx(set_ctx2); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   58|  1.97M|    ((union alias16 *) &(var)[off])->u16 = (val) * 0x0101
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|  1.97M|                txtp_map += 32; \
  |  |  |  |  799|  1.97M|            }
  |  |  ------------------
  |  |  |  Branch (82:5): [True: 790k, False: 5.57M]
  |  |  ------------------
  |  |   83|   690k|    case 2: set_ctx(set_ctx4); break; \
  |  |  ------------------
  |  |  |  |  796|  2.98M|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 2.29M, False: 690k]
  |  |  |  |  ------------------
  |  |  |  |  797|  2.29M|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   83|  2.29M|    case 2: set_ctx(set_ctx4); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   60|  2.29M|    ((union alias32 *) &(var)[off])->u32 = (val) * 0x01010101U
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|  2.29M|                txtp_map += 32; \
  |  |  |  |  799|  2.29M|            }
  |  |  ------------------
  |  |  |  Branch (83:5): [True: 690k, False: 5.67M]
  |  |  ------------------
  |  |   84|   226k|    case 3: set_ctx(set_ctx8); break; \
  |  |  ------------------
  |  |  |  |  796|  1.54M|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 1.32M, False: 226k]
  |  |  |  |  ------------------
  |  |  |  |  797|  1.32M|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  1.32M|    case 3: set_ctx(set_ctx8); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   62|  1.32M|    ((union alias64 *) &(var)[off])->u64 = (val) * 0x0101010101010101ULL
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|  1.32M|                txtp_map += 32; \
  |  |  |  |  799|  1.32M|            }
  |  |  ------------------
  |  |  |  Branch (84:5): [True: 226k, False: 6.13M]
  |  |  ------------------
  |  |   85|   164k|    case 4: set_ctx(set_ctx16); break; \
  |  |  ------------------
  |  |  |  |  796|  2.45M|            for (int y = 0; y < txh; y++) { \
  |  |  |  |  ------------------
  |  |  |  |  |  Branch (796:29): [True: 2.28M, False: 164k]
  |  |  |  |  ------------------
  |  |  |  |  797|  2.28M|                rep_macro(txtp_map, 0, txtp); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  2.28M|    case 4: set_ctx(set_ctx16); break; \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   63|  2.28M|#define set_ctx16(var, off, val) do { \
  |  |  |  |  |  |  |  |   64|  2.28M|        memset(&(var)[off], val, 16); \
  |  |  |  |  |  |  |  |   65|  2.28M|    } while (0)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  Branch (65:14): [Folded, False: 2.28M]
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  798|  2.28M|                txtp_map += 32; \
  |  |  |  |  799|  2.28M|            }
  |  |  ------------------
  |  |  |  Branch (85:5): [True: 164k, False: 6.19M]
  |  |  ------------------
  |  |   86|      0|    default: assert(0); \
  |  |  ------------------
  |  |  |  Branch (86:5): [True: 0, False: 6.36M]
  |  |  ------------------
  |  |   87|  6.36M|    }
  ------------------
  |  Branch (801:13): [Folded, False: 0]
  ------------------
  802|  6.36M|#undef set_ctx
  803|  6.36M|            if (t->frame_thread.pass == 1)
  ------------------
  |  Branch (803:17): [True: 6.36M, False: 66]
  ------------------
  804|  6.36M|                *ts->frame_thread[1].cbi++ = eob * (1 << 5) + txtp;
  805|  6.36M|        } else {
  806|   859k|            const int cbi = *ts->frame_thread[0].cbi++;
  807|   859k|            eob  = cbi >> 5;
  808|   859k|            txtp = cbi & 0x1f;
  809|   859k|        }
  810|  7.22M|        if (!(t->frame_thread.pass & 1)) {
  ------------------
  |  Branch (810:13): [True: 868k, False: 6.35M]
  ------------------
  811|   868k|            assert(dst);
  ------------------
  |  Branch (811:13): [True: 868k, False: 18.4E]
  ------------------
  812|   868k|            if (eob >= 0) {
  ------------------
  |  Branch (812:17): [True: 700k, False: 168k]
  ------------------
  813|   700k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   700k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 700k]
  |  |  ------------------
  |  |   35|   700k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   700k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
  814|      0|                    coef_dump(cf, imin(t_dim->h, 8) * 4, imin(t_dim->w, 8) * 4, 3, "dq");
  815|   700k|                dsp->itx.itxfm_add[ytx][txtp](dst, f->cur.stride[0], cf, eob
  816|   700k|                                              HIGHBD_CALL_SUFFIX);
  817|   700k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   700k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 700k]
  |  |  ------------------
  |  |   35|   700k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   700k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
  818|      0|                    hex_dump(dst, f->cur.stride[0], t_dim->w * 4, t_dim->h * 4, "recon");
  819|   700k|            }
  820|   868k|        }
  821|  7.22M|    }
  822|  7.57M|}
recon_tmpl.c:decode_coefs:
  327|  49.9M|{
  328|  49.9M|    Dav1dTileState *const ts = t->ts;
  329|  49.9M|    const int chroma = !!plane;
  330|  49.9M|    const Dav1dFrameContext *const f = t->f;
  331|  49.9M|    const int lossless = f->frame_hdr->segmentation.lossless[b->seg_id];
  332|  49.9M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[tx];
  333|  49.9M|    const int dbg = DEBUG_BLOCK_INFO && plane && 0;
  ------------------
  |  |   34|  49.9M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 49.9M]
  |  |  ------------------
  |  |   35|  49.9M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  49.9M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  |  Branch (333:41): [True: 0, False: 0]
  |  Branch (333:50): [Folded, False: 0]
  ------------------
  334|       |
  335|  49.9M|    if (dbg)
  ------------------
  |  Branch (335:9): [Folded, False: 49.9M]
  ------------------
  336|      0|        printf("Start: r=%d\n", ts->msac.rng);
  337|       |
  338|       |    // does this block have any non-zero coefficients
  339|  49.9M|    const int sctx = get_skip_ctx(t_dim, bs, a, l, chroma, f->cur.p.layout);
  340|  49.9M|    const int all_skip = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|  49.9M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  341|  49.9M|                             ts->cdf.coef.skip[t_dim->ctx][sctx]);
  342|  49.9M|    if (dbg)
  ------------------
  |  Branch (342:9): [Folded, False: 49.9M]
  ------------------
  343|      0|        printf("Post-non-zero[%d][%d][%d]: r=%d\n",
  344|      0|               t_dim->ctx, sctx, all_skip, ts->msac.rng);
  345|  49.9M|    if (all_skip) {
  ------------------
  |  Branch (345:9): [True: 32.2M, False: 17.7M]
  ------------------
  346|  32.2M|        *res_ctx = 0x40;
  347|  32.2M|        *txtp = lossless * WHT_WHT; /* lossless ? WHT_WHT : DCT_DCT */
  348|  32.2M|        return -1;
  349|  32.2M|    }
  350|       |
  351|       |    // transform type (chroma: derived, luma: explicitly coded)
  352|  17.7M|    if (lossless) {
  ------------------
  |  Branch (352:9): [True: 7.60M, False: 10.1M]
  ------------------
  353|  7.60M|        assert(t_dim->max == TX_4X4);
  ------------------
  |  Branch (353:9): [True: 7.60M, False: 18.4E]
  ------------------
  354|  7.60M|        *txtp = WHT_WHT;
  355|  10.1M|    } else if (t_dim->max + intra >= TX_64X64) {
  ------------------
  |  Branch (355:16): [True: 2.72M, False: 7.45M]
  ------------------
  356|  2.72M|        *txtp = DCT_DCT;
  357|  7.45M|    } else if (chroma) {
  ------------------
  |  Branch (357:16): [True: 2.50M, False: 4.95M]
  ------------------
  358|       |        // inferred from either the luma txtp (inter) or a LUT (intra)
  359|  2.50M|        *txtp = intra ? dav1d_txtp_from_uvmode[b->uv_mode] :
  ------------------
  |  Branch (359:17): [True: 1.65M, False: 851k]
  ------------------
  360|  2.50M|                        get_uv_inter_txtp(t_dim, *txtp);
  361|  4.95M|    } else if (!f->frame_hdr->segmentation.qidx[b->seg_id]) {
  ------------------
  |  Branch (361:16): [True: 6.60k, False: 4.94M]
  ------------------
  362|       |        // In libaom, lossless is checked by a literal qidx == 0, but not all
  363|       |        // such blocks are actually lossless. The remainder gets an implicit
  364|       |        // transform type (for luma)
  365|  6.60k|        *txtp = DCT_DCT;
  366|  4.94M|    } else {
  367|  4.94M|        unsigned idx;
  368|  4.94M|        if (intra) {
  ------------------
  |  Branch (368:13): [True: 3.49M, False: 1.44M]
  ------------------
  369|  3.49M|            const enum IntraPredMode y_mode_nofilt = b->y_mode == FILTER_PRED ?
  ------------------
  |  Branch (369:54): [True: 663k, False: 2.83M]
  ------------------
  370|  2.83M|                dav1d_filter_mode_to_y_mode[b->y_angle] : b->y_mode;
  371|  3.49M|            if (f->frame_hdr->reduced_txtp_set || t_dim->min == TX_16X16) {
  ------------------
  |  Branch (371:17): [True: 547k, False: 2.94M]
  |  Branch (371:51): [True: 471k, False: 2.47M]
  ------------------
  372|  1.02M|                idx = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  1.02M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  373|  1.02M|                          ts->cdf.m.txtp_intra2[t_dim->min][y_mode_nofilt], 4);
  374|  1.02M|                *txtp = dav1d_tx_types_per_set[idx + 0];
  375|  2.47M|            } else {
  376|  2.47M|                idx = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  2.47M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
  377|  2.47M|                          ts->cdf.m.txtp_intra1[t_dim->min][y_mode_nofilt], 6);
  378|  2.47M|                *txtp = dav1d_tx_types_per_set[idx + 5];
  379|  2.47M|            }
  380|  3.49M|            if (dbg)
  ------------------
  |  Branch (380:17): [Folded, False: 3.49M]
  ------------------
  381|      0|                printf("Post-txtp-intra[%d->%d][%d][%d->%d]: r=%d\n",
  382|      0|                       tx, t_dim->min, y_mode_nofilt, idx, *txtp, ts->msac.rng);
  383|  3.49M|        } else {
  384|  1.75M|            if (f->frame_hdr->reduced_txtp_set || t_dim->max == TX_32X32) {
  ------------------
  |  Branch (384:17): [True: 18.4E, False: 1.75M]
  |  Branch (384:51): [True: 267k, False: 1.49M]
  ------------------
  385|   379k|                idx = dav1d_msac_decode_bool_adapt(&ts->msac,
  ------------------
  |  |   52|   379k|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  386|   379k|                          ts->cdf.m.txtp_inter3[t_dim->min]);
  387|   379k|                *txtp = (idx - 1) & IDTX; /* idx ? DCT_DCT : IDTX */
  388|  1.07M|            } else if (t_dim->min == TX_16X16) {
  ------------------
  |  Branch (388:24): [True: 215k, False: 854k]
  ------------------
  389|   215k|                idx = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|   215k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
  390|   215k|                          ts->cdf.m.txtp_inter2, 11);
  391|   215k|                *txtp = dav1d_tx_types_per_set[idx + 12];
  392|   854k|            } else {
  393|   854k|                idx = dav1d_msac_decode_symbol_adapt16(&ts->msac,
  ------------------
  |  |   57|   854k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  ------------------
  394|   854k|                          ts->cdf.m.txtp_inter1[t_dim->min], 15);
  395|   854k|                *txtp = dav1d_tx_types_per_set[idx + 24];
  396|   854k|            }
  397|  1.44M|            if (dbg)
  ------------------
  |  Branch (397:17): [Folded, False: 1.44M]
  ------------------
  398|      0|                printf("Post-txtp-inter[%d->%d][%d->%d]: r=%d\n",
  399|      0|                       tx, t_dim->min, idx, *txtp, ts->msac.rng);
  400|  1.44M|        }
  401|  4.94M|    }
  402|       |
  403|       |    // find end-of-block (eob)
  404|  17.7M|    int eob;
  405|  17.7M|    const int slw = imin(t_dim->lw, TX_32X32), slh = imin(t_dim->lh, TX_32X32);
  406|  17.7M|    const int tx2dszctx = slw + slh;
  407|  17.7M|    const enum TxClass tx_class = dav1d_tx_type_class[*txtp];
  408|  17.7M|    const int is_1d = tx_class != TX_CLASS_2D;
  409|  17.7M|    switch (tx2dszctx) {
  ------------------
  |  Branch (409:13): [True: 18.2M, False: 18.4E]
  ------------------
  410|      0|#define case_sz(sz, bin, ns, is_1d) \
  411|      0|    case sz: { \
  412|      0|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  413|      0|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  414|      0|        break; \
  415|      0|    }
  416|  8.89M|    case_sz(0,   16,  8, [is_1d]);
  ------------------
  |  |  411|  8.89M|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 8.89M, False: 8.89M]
  |  |  ------------------
  |  |  412|  8.89M|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|  8.89M|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   48|  8.89M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  |  |  ------------------
  |  |  414|  8.89M|        break; \
  |  |  415|  8.89M|    }
  ------------------
  417|  1.28M|    case_sz(1,   32,  8, [is_1d]);
  ------------------
  |  |  411|  1.28M|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 1.28M, False: 16.4M]
  |  |  ------------------
  |  |  412|  1.28M|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|  1.28M|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   48|  1.28M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  |  |  ------------------
  |  |  414|  1.28M|        break; \
  |  |  415|  1.28M|    }
  ------------------
  418|  2.49M|    case_sz(2,   64,  8, [is_1d]);
  ------------------
  |  |  411|  2.49M|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 2.49M, False: 15.2M]
  |  |  ------------------
  |  |  412|  2.49M|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|  2.49M|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   48|  2.49M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  |  |  ------------------
  |  |  414|  2.49M|        break; \
  |  |  415|  2.49M|    }
  ------------------
  419|  1.32M|    case_sz(3,  128,  8, [is_1d]);
  ------------------
  |  |  411|  1.32M|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 1.32M, False: 16.4M]
  |  |  ------------------
  |  |  412|  1.32M|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|  1.32M|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   48|  1.32M|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  |  |  ------------------
  |  |  414|  1.32M|        break; \
  |  |  415|  1.32M|    }
  ------------------
  420|  1.59M|    case_sz(4,  256, 16, [is_1d]);
  ------------------
  |  |  411|  1.59M|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 1.59M, False: 16.1M]
  |  |  ------------------
  |  |  412|  1.59M|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|  1.59M|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   57|  1.59M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  |  |  ------------------
  |  |  414|  1.59M|        break; \
  |  |  415|  1.59M|    }
  ------------------
  421|   588k|    case_sz(5,  512, 16,        );
  ------------------
  |  |  411|   588k|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 588k, False: 17.1M]
  |  |  ------------------
  |  |  412|   588k|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|   588k|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   57|   588k|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  |  |  ------------------
  |  |  414|   588k|        break; \
  |  |  415|   588k|    }
  ------------------
  422|  2.01M|    case_sz(6, 1024, 16,        );
  ------------------
  |  |  411|  2.01M|    case sz: { \
  |  |  ------------------
  |  |  |  Branch (411:5): [True: 2.01M, False: 15.7M]
  |  |  ------------------
  |  |  412|  2.01M|        uint16_t *const eob_bin_cdf = ts->cdf.coef.eob_bin_##bin[chroma]is_1d; \
  |  |  413|  2.01M|        eob = dav1d_msac_decode_symbol_adapt##ns(&ts->msac, eob_bin_cdf, 4 + sz); \
  |  |  ------------------
  |  |  |  |   57|  2.01M|#define dav1d_msac_decode_symbol_adapt16(ctx, cdf, symb) ((ctx)->symbol_adapt16(ctx, cdf, symb))
  |  |  ------------------
  |  |  414|  2.01M|        break; \
  |  |  415|  2.01M|    }
  ------------------
  423|  17.7M|#undef case_sz
  424|  17.7M|    }
  425|  18.1M|    if (dbg)
  ------------------
  |  Branch (425:9): [Folded, False: 18.1M]
  ------------------
  426|      0|        printf("Post-eob_bin_%d[%d][%d][%d]: r=%d\n",
  427|      0|               16 << tx2dszctx, chroma, is_1d, eob, ts->msac.rng);
  428|  18.1M|    if (eob > 1) {
  ------------------
  |  Branch (428:9): [True: 13.0M, False: 5.14M]
  ------------------
  429|  13.0M|        const int eob_bin = eob - 2;
  430|  13.0M|        uint16_t *const eob_hi_bit_cdf =
  431|  13.0M|            ts->cdf.coef.eob_hi_bit[t_dim->ctx][chroma][eob_bin];
  432|  13.0M|        const int eob_hi_bit = dav1d_msac_decode_bool_adapt(&ts->msac, eob_hi_bit_cdf);
  ------------------
  |  |   52|  13.0M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  433|  13.0M|        if (dbg)
  ------------------
  |  Branch (433:13): [Folded, False: 13.0M]
  ------------------
  434|      0|            printf("Post-eob_hi_bit[%d][%d][%d][%d]: r=%d\n",
  435|      0|                   t_dim->ctx, chroma, eob_bin, eob_hi_bit, ts->msac.rng);
  436|  13.0M|        eob = ((eob_hi_bit | 2) << eob_bin) | dav1d_msac_decode_bools(&ts->msac, eob_bin);
  437|  13.0M|        if (dbg)
  ------------------
  |  Branch (437:13): [Folded, False: 13.0M]
  ------------------
  438|      0|            printf("Post-eob[%d]: r=%d\n", eob, ts->msac.rng);
  439|  13.0M|    }
  440|  18.1M|    assert(eob >= 0);
  ------------------
  |  Branch (440:5): [True: 18.1M, False: 18.4E]
  ------------------
  441|       |
  442|       |    // base tokens
  443|  18.1M|    uint16_t (*const eob_cdf)[4] = ts->cdf.coef.eob_base_tok[t_dim->ctx][chroma];
  444|  18.1M|    uint16_t (*const hi_cdf)[4] = ts->cdf.coef.br_tok[imin(t_dim->ctx, 3)][chroma];
  445|  18.1M|    unsigned rc, dc_tok;
  446|       |
  447|  18.1M|    if (eob) {
  ------------------
  |  Branch (447:9): [True: 13.6M, False: 4.46M]
  ------------------
  448|  13.6M|        uint16_t (*const lo_cdf)[4] = ts->cdf.coef.base_tok[t_dim->ctx][chroma];
  449|  13.6M|        uint8_t *const levels = t->scratch.levels; // bits 0-5: tok, 6-7: lo_tok
  450|       |
  451|       |        /* eob */
  452|  13.6M|        unsigned ctx = 1 + (eob > 2 << tx2dszctx) + (eob > 4 << tx2dszctx);
  453|  13.6M|        int eob_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, eob_cdf[ctx], 2);
  ------------------
  |  |   47|  13.6M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  454|  13.6M|        int tok = eob_tok + 1;
  455|  13.6M|        int level_tok = tok * 0x41;
  456|  13.6M|        unsigned mag;
  457|       |
  458|  13.6M|#define DECODE_COEFS_CLASS(tx_class) \
  459|  13.6M|        unsigned x, y; \
  460|  13.6M|        uint8_t *level; \
  461|  13.6M|        if (tx_class == TX_CLASS_2D) \
  462|  13.6M|            rc = scan[eob], x = rc >> shift, y = rc & mask; \
  463|  13.6M|        else if (tx_class == TX_CLASS_H) \
  464|       |            /* Transposing reduces the stride and padding requirements */ \
  465|  13.6M|            x = eob & mask, y = eob >> shift, rc = eob; \
  466|  13.6M|        else /* tx_class == TX_CLASS_V */ \
  467|  13.6M|            x = eob & mask, y = eob >> shift, rc = (x << shift2) | y; \
  468|  13.6M|        if (dbg) \
  469|  13.6M|            printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  470|  13.6M|                   t_dim->ctx, chroma, ctx, eob, rc, tok, ts->msac.rng); \
  471|  13.6M|        if (eob_tok == 2) { \
  472|  13.6M|            ctx = (tx_class == TX_CLASS_2D ? (x | y) > 1 : y != 0) ? 14 : 7; \
  473|  13.6M|            tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  474|  13.6M|            level_tok = tok + (3 << 6); \
  475|  13.6M|            if (dbg) \
  476|  13.6M|                printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  477|  13.6M|                       imin(t_dim->ctx, 3), chroma, ctx, eob, rc, tok, \
  478|  13.6M|                       ts->msac.rng); \
  479|  13.6M|        } \
  480|  13.6M|        cf[rc] = tok << 11; \
  481|  13.6M|        if (tx_class == TX_CLASS_2D) \
  482|  13.6M|            level = levels + rc; \
  483|  13.6M|        else \
  484|  13.6M|            level = levels + x * stride + y; \
  485|  13.6M|        *level = (uint8_t) level_tok; \
  486|  13.6M|        for (int i = eob - 1; i > 0; i--) { /* ac */ \
  487|  13.6M|            unsigned rc_i; \
  488|  13.6M|            if (tx_class == TX_CLASS_2D) \
  489|  13.6M|                rc_i = scan[i], x = rc_i >> shift, y = rc_i & mask; \
  490|  13.6M|            else if (tx_class == TX_CLASS_H) \
  491|  13.6M|                x = i & mask, y = i >> shift, rc_i = i; \
  492|  13.6M|            else /* tx_class == TX_CLASS_V */ \
  493|  13.6M|                x = i & mask, y = i >> shift, rc_i = (x << shift2) | y; \
  494|  13.6M|            assert(x < 32 && y < 32); \
  495|  13.6M|            if (tx_class == TX_CLASS_2D) \
  496|  13.6M|                level = levels + rc_i; \
  497|  13.6M|            else \
  498|  13.6M|                level = levels + x * stride + y; \
  499|  13.6M|            ctx = get_lo_ctx(level, tx_class, &mag, lo_ctx_offsets, x, y, stride); \
  500|  13.6M|            if (tx_class == TX_CLASS_2D) \
  501|  13.6M|                y |= x; \
  502|  13.6M|            tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  503|  13.6M|            if (dbg) \
  504|  13.6M|                printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  505|  13.6M|                       t_dim->ctx, chroma, ctx, i, rc_i, tok, ts->msac.rng); \
  506|  13.6M|            if (tok == 3) { \
  507|  13.6M|                mag &= 63; \
  508|  13.6M|                ctx = (y > (tx_class == TX_CLASS_2D) ? 14 : 7) + \
  509|  13.6M|                      (mag > 12 ? 6 : (mag + 1) >> 1); \
  510|  13.6M|                tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  511|  13.6M|                if (dbg) \
  512|  13.6M|                    printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  513|  13.6M|                           imin(t_dim->ctx, 3), chroma, ctx, i, rc_i, tok, \
  514|  13.6M|                           ts->msac.rng); \
  515|  13.6M|                *level = (uint8_t) (tok + (3 << 6)); \
  516|  13.6M|                cf[rc_i] = (tok << 11) | rc; \
  517|  13.6M|                rc = rc_i; \
  518|  13.6M|            } else { \
  519|       |                /* 0x1 for tok, 0x7ff as bitmask for rc, 0x41 for level_tok */ \
  520|  13.6M|                tok *= 0x17ff41; \
  521|  13.6M|                *level = (uint8_t) tok; \
  522|       |                /* tok ? (tok << 11) | rc : 0 */ \
  523|  13.6M|                tok = (tok >> 9) & (rc + ~0x7ffu); \
  524|  13.6M|                if (tok) rc = rc_i; \
  525|  13.6M|                cf[rc_i] = tok; \
  526|  13.6M|            } \
  527|  13.6M|        } \
  528|       |        /* dc */ \
  529|  13.6M|        ctx = (tx_class == TX_CLASS_2D) ? 0 : \
  530|  13.6M|            get_lo_ctx(levels, tx_class, &mag, lo_ctx_offsets, 0, 0, stride); \
  531|  13.6M|        dc_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  532|  13.6M|        if (dbg) \
  533|  13.6M|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n", \
  534|  13.6M|                   t_dim->ctx, chroma, ctx, dc_tok, ts->msac.rng); \
  535|  13.6M|        if (dc_tok == 3) { \
  536|  13.6M|            if (tx_class == TX_CLASS_2D) \
  537|  13.6M|                mag = levels[0 * stride + 1] + levels[1 * stride + 0] + \
  538|  13.6M|                      levels[1 * stride + 1]; \
  539|  13.6M|            mag &= 63; \
  540|  13.6M|            ctx = mag > 12 ? 6 : (mag + 1) >> 1; \
  541|  13.6M|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  542|  13.6M|            if (dbg) \
  543|  13.6M|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n", \
  544|  13.6M|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng); \
  545|  13.6M|        } \
  546|  13.6M|        break
  547|       |
  548|  13.6M|        const uint16_t *scan;
  549|  13.6M|        switch (tx_class) {
  550|  12.8M|        case TX_CLASS_2D: {
  ------------------
  |  Branch (550:9): [True: 12.8M, False: 827k]
  ------------------
  551|  12.8M|            const unsigned nonsquare_tx = tx >= RTX_4X8;
  552|  12.8M|            const uint8_t (*const lo_ctx_offsets)[5] =
  553|  12.8M|                dav1d_lo_ctx_offsets[nonsquare_tx + (tx & nonsquare_tx)];
  554|  12.8M|            scan = dav1d_scans[tx];
  555|  12.8M|            const ptrdiff_t stride = 4 << slh;
  556|  12.8M|            const unsigned shift = slh + 2, shift2 = 0;
  557|  12.8M|            const unsigned mask = (4 << slh) - 1;
  558|  12.8M|            memset(levels, 0, stride * ((4 << slw) + 2));
  559|  12.8M|            DECODE_COEFS_CLASS(TX_CLASS_2D);
  ------------------
  |  |  459|  12.8M|        unsigned x, y; \
  |  |  460|  12.8M|        uint8_t *level; \
  |  |  461|  12.8M|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (461:13): [True: 12.8M, Folded]
  |  |  ------------------
  |  |  462|  12.8M|            rc = scan[eob], x = rc >> shift, y = rc & mask; \
  |  |  463|  12.8M|        else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (463:18): [Folded, False: 8.26k]
  |  |  ------------------
  |  |  464|  8.26k|            /* Transposing reduces the stride and padding requirements */ \
  |  |  465|  8.26k|            x = eob & mask, y = eob >> shift, rc = eob; \
  |  |  466|  8.26k|        else /* tx_class == TX_CLASS_V */ \
  |  |  467|  8.26k|            x = eob & mask, y = eob >> shift, rc = (x << shift2) | y; \
  |  |  468|  12.8M|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (468:13): [Folded, False: 12.8M]
  |  |  ------------------
  |  |  469|  12.8M|            printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  470|      0|                   t_dim->ctx, chroma, ctx, eob, rc, tok, ts->msac.rng); \
  |  |  471|  12.8M|        if (eob_tok == 2) { \
  |  |  ------------------
  |  |  |  Branch (471:13): [True: 447k, False: 12.4M]
  |  |  ------------------
  |  |  472|  18.4E|            ctx = (tx_class == TX_CLASS_2D ? (x | y) > 1 : y != 0) ? 14 : 7; \
  |  |  ------------------
  |  |  |  Branch (472:19): [True: 442k, False: 4.48k]
  |  |  |  Branch (472:20): [True: 447k, Folded]
  |  |  ------------------
  |  |  473|   447k|            tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|   447k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  474|   447k|            level_tok = tok + (3 << 6); \
  |  |  475|   447k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (475:17): [Folded, False: 447k]
  |  |  ------------------
  |  |  476|   447k|                printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  477|      0|                       imin(t_dim->ctx, 3), chroma, ctx, eob, rc, tok, \
  |  |  478|      0|                       ts->msac.rng); \
  |  |  479|   447k|        } \
  |  |  480|  12.8M|        cf[rc] = tok << 11; \
  |  |  481|  12.8M|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (481:13): [True: 12.8M, Folded]
  |  |  ------------------
  |  |  482|  12.8M|            level = levels + rc; \
  |  |  483|  12.8M|        else \
  |  |  484|  12.8M|            level = levels + x * stride + y; \
  |  |  485|  12.8M|        *level = (uint8_t) level_tok; \
  |  |  486|   219M|        for (int i = eob - 1; i > 0; i--) { /* ac */ \
  |  |  ------------------
  |  |  |  Branch (486:31): [True: 206M, False: 13.3M]
  |  |  ------------------
  |  |  487|   206M|            unsigned rc_i; \
  |  |  488|   206M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (488:17): [True: 206M, Folded]
  |  |  ------------------
  |  |  489|   206M|                rc_i = scan[i], x = rc_i >> shift, y = rc_i & mask; \
  |  |  490|   206M|            else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (490:22): [Folded, False: 103k]
  |  |  ------------------
  |  |  491|   103k|                x = i & mask, y = i >> shift, rc_i = i; \
  |  |  492|   103k|            else /* tx_class == TX_CLASS_V */ \
  |  |  493|   103k|                x = i & mask, y = i >> shift, rc_i = (x << shift2) | y; \
  |  |  494|   206M|            assert(x < 32 && y < 32); \
  |  |  495|   206M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (495:17): [True: 206M, Folded]
  |  |  ------------------
  |  |  496|   206M|                level = levels + rc_i; \
  |  |  497|   206M|            else \
  |  |  498|  18.4E|                level = levels + x * stride + y; \
  |  |  499|   206M|            ctx = get_lo_ctx(level, tx_class, &mag, lo_ctx_offsets, x, y, stride); \
  |  |  500|   206M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (500:17): [True: 204M, Folded]
  |  |  ------------------
  |  |  501|   206M|                y |= x; \
  |  |  502|   206M|            tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|   206M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  503|   206M|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (503:17): [Folded, False: 206M]
  |  |  ------------------
  |  |  504|   206M|                printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  505|      0|                       t_dim->ctx, chroma, ctx, i, rc_i, tok, ts->msac.rng); \
  |  |  506|   206M|            if (tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (506:17): [True: 19.2M, False: 187M]
  |  |  ------------------
  |  |  507|  19.2M|                mag &= 63; \
  |  |  508|  19.2M|                ctx = (y > (tx_class == TX_CLASS_2D) ? 14 : 7) + \
  |  |  ------------------
  |  |  |  Branch (508:24): [True: 13.1M, False: 6.09M]
  |  |  ------------------
  |  |  509|  19.2M|                      (mag > 12 ? 6 : (mag + 1) >> 1); \
  |  |  ------------------
  |  |  |  Branch (509:24): [True: 3.39M, False: 15.8M]
  |  |  ------------------
  |  |  510|  19.2M|                tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  19.2M|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  511|  19.2M|                if (dbg) \
  |  |  ------------------
  |  |  |  Branch (511:21): [Folded, False: 19.2M]
  |  |  ------------------
  |  |  512|  19.2M|                    printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  513|      0|                           imin(t_dim->ctx, 3), chroma, ctx, i, rc_i, tok, \
  |  |  514|      0|                           ts->msac.rng); \
  |  |  515|  19.2M|                *level = (uint8_t) (tok + (3 << 6)); \
  |  |  516|  19.2M|                cf[rc_i] = (tok << 11) | rc; \
  |  |  517|  19.2M|                rc = rc_i; \
  |  |  518|   187M|            } else { \
  |  |  519|   187M|                /* 0x1 for tok, 0x7ff as bitmask for rc, 0x41 for level_tok */ \
  |  |  520|   187M|                tok *= 0x17ff41; \
  |  |  521|   187M|                *level = (uint8_t) tok; \
  |  |  522|   187M|                /* tok ? (tok << 11) | rc : 0 */ \
  |  |  523|   187M|                tok = (tok >> 9) & (rc + ~0x7ffu); \
  |  |  524|   187M|                if (tok) rc = rc_i; \
  |  |  ------------------
  |  |  |  Branch (524:21): [True: 67.6M, False: 119M]
  |  |  ------------------
  |  |  525|   187M|                cf[rc_i] = tok; \
  |  |  526|   187M|            } \
  |  |  527|   206M|        } \
  |  |  528|  12.8M|        /* dc */ \
  |  |  529|  13.3M|        ctx = (tx_class == TX_CLASS_2D) ? 0 : \
  |  |  ------------------
  |  |  |  Branch (529:15): [True: 12.8M, Folded]
  |  |  ------------------
  |  |  530|  13.3M|            get_lo_ctx(levels, tx_class, &mag, lo_ctx_offsets, 0, 0, stride); \
  |  |  531|  13.3M|        dc_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|  13.3M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  532|  13.3M|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (532:13): [Folded, False: 13.3M]
  |  |  ------------------
  |  |  533|  13.3M|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n", \
  |  |  534|      0|                   t_dim->ctx, chroma, ctx, dc_tok, ts->msac.rng); \
  |  |  535|  13.3M|        if (dc_tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (535:13): [True: 5.31M, False: 8.00M]
  |  |  ------------------
  |  |  536|  5.31M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (536:17): [True: 5.31M, Folded]
  |  |  ------------------
  |  |  537|  5.31M|                mag = levels[0 * stride + 1] + levels[1 * stride + 0] + \
  |  |  538|  5.31M|                      levels[1 * stride + 1]; \
  |  |  539|  5.31M|            mag &= 63; \
  |  |  540|  5.31M|            ctx = mag > 12 ? 6 : (mag + 1) >> 1; \
  |  |  ------------------
  |  |  |  Branch (540:19): [True: 806k, False: 4.50M]
  |  |  ------------------
  |  |  541|  5.31M|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  5.31M|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  542|  5.31M|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (542:17): [Folded, False: 5.31M]
  |  |  ------------------
  |  |  543|  5.31M|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n", \
  |  |  544|      0|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng); \
  |  |  545|  5.31M|        } \
  |  |  546|  13.3M|        break
  ------------------
  |  Branch (559:13): [True: 206M, False: 18.4E]
  |  Branch (559:13): [True: 206M, False: 49.3k]
  ------------------
  560|  12.8M|        }
  561|   541k|        case TX_CLASS_H: {
  ------------------
  |  Branch (561:9): [True: 541k, False: 13.1M]
  ------------------
  562|   541k|            const uint8_t (*const lo_ctx_offsets)[5] = NULL;
  563|   541k|            const ptrdiff_t stride = 16;
  564|   541k|            const unsigned shift = slh + 2, shift2 = 0;
  565|   541k|            const unsigned mask = (4 << slh) - 1;
  566|   541k|            memset(levels, 0, stride * ((4 << slh) + 2));
  567|   541k|            DECODE_COEFS_CLASS(TX_CLASS_H);
  ------------------
  |  |  459|   541k|        unsigned x, y; \
  |  |  460|   541k|        uint8_t *level; \
  |  |  461|   541k|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (461:13): [Folded, False: 541k]
  |  |  ------------------
  |  |  462|   541k|            rc = scan[eob], x = rc >> shift, y = rc & mask; \
  |  |  463|   541k|        else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (463:18): [True: 541k, Folded]
  |  |  ------------------
  |  |  464|   541k|            /* Transposing reduces the stride and padding requirements */ \
  |  |  465|   541k|            x = eob & mask, y = eob >> shift, rc = eob; \
  |  |  466|   541k|        else /* tx_class == TX_CLASS_V */ \
  |  |  467|   541k|            x = eob & mask, y = eob >> shift, rc = (x << shift2) | y; \
  |  |  468|   541k|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (468:13): [Folded, False: 541k]
  |  |  ------------------
  |  |  469|   541k|            printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  470|      0|                   t_dim->ctx, chroma, ctx, eob, rc, tok, ts->msac.rng); \
  |  |  471|   541k|        if (eob_tok == 2) { \
  |  |  ------------------
  |  |  |  Branch (471:13): [True: 11.3k, False: 530k]
  |  |  ------------------
  |  |  472|  11.3k|            ctx = (tx_class == TX_CLASS_2D ? (x | y) > 1 : y != 0) ? 14 : 7; \
  |  |  ------------------
  |  |  |  Branch (472:19): [True: 8.56k, False: 2.77k]
  |  |  |  Branch (472:20): [Folded, False: 11.3k]
  |  |  ------------------
  |  |  473|  11.3k|            tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  11.3k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  474|  11.3k|            level_tok = tok + (3 << 6); \
  |  |  475|  11.3k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (475:17): [Folded, False: 11.3k]
  |  |  ------------------
  |  |  476|  11.3k|                printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  477|      0|                       imin(t_dim->ctx, 3), chroma, ctx, eob, rc, tok, \
  |  |  478|      0|                       ts->msac.rng); \
  |  |  479|  11.3k|        } \
  |  |  480|   541k|        cf[rc] = tok << 11; \
  |  |  481|   541k|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (481:13): [Folded, False: 541k]
  |  |  ------------------
  |  |  482|   541k|            level = levels + rc; \
  |  |  483|   541k|        else \
  |  |  484|   541k|            level = levels + x * stride + y; \
  |  |  485|   541k|        *level = (uint8_t) level_tok; \
  |  |  486|  10.6M|        for (int i = eob - 1; i > 0; i--) { /* ac */ \
  |  |  ------------------
  |  |  |  Branch (486:31): [True: 10.1M, False: 540k]
  |  |  ------------------
  |  |  487|  10.1M|            unsigned rc_i; \
  |  |  488|  10.1M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (488:17): [Folded, False: 10.1M]
  |  |  ------------------
  |  |  489|  10.1M|                rc_i = scan[i], x = rc_i >> shift, y = rc_i & mask; \
  |  |  490|  10.1M|            else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (490:22): [True: 10.1M, Folded]
  |  |  ------------------
  |  |  491|  10.1M|                x = i & mask, y = i >> shift, rc_i = i; \
  |  |  492|  10.1M|            else /* tx_class == TX_CLASS_V */ \
  |  |  493|  10.1M|                x = i & mask, y = i >> shift, rc_i = (x << shift2) | y; \
  |  |  494|  10.1M|            assert(x < 32 && y < 32); \
  |  |  495|  10.1M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (495:17): [Folded, False: 10.1M]
  |  |  ------------------
  |  |  496|  10.1M|                level = levels + rc_i; \
  |  |  497|  10.1M|            else \
  |  |  498|  10.1M|                level = levels + x * stride + y; \
  |  |  499|  10.1M|            ctx = get_lo_ctx(level, tx_class, &mag, lo_ctx_offsets, x, y, stride); \
  |  |  500|  10.1M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (500:17): [Folded, False: 10.1M]
  |  |  ------------------
  |  |  501|  10.1M|                y |= x; \
  |  |  502|  10.1M|            tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|  10.1M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  503|  10.1M|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (503:17): [Folded, False: 10.1M]
  |  |  ------------------
  |  |  504|  10.1M|                printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  505|      0|                       t_dim->ctx, chroma, ctx, i, rc_i, tok, ts->msac.rng); \
  |  |  506|  10.1M|            if (tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (506:17): [True: 513k, False: 9.64M]
  |  |  ------------------
  |  |  507|   513k|                mag &= 63; \
  |  |  508|   513k|                ctx = (y > (tx_class == TX_CLASS_2D) ? 14 : 7) + \
  |  |  ------------------
  |  |  |  Branch (508:24): [True: 314k, False: 198k]
  |  |  ------------------
  |  |  509|   513k|                      (mag > 12 ? 6 : (mag + 1) >> 1); \
  |  |  ------------------
  |  |  |  Branch (509:24): [True: 46.8k, False: 466k]
  |  |  ------------------
  |  |  510|   513k|                tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|   513k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  511|   513k|                if (dbg) \
  |  |  ------------------
  |  |  |  Branch (511:21): [Folded, False: 513k]
  |  |  ------------------
  |  |  512|   513k|                    printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  513|      0|                           imin(t_dim->ctx, 3), chroma, ctx, i, rc_i, tok, \
  |  |  514|      0|                           ts->msac.rng); \
  |  |  515|   513k|                *level = (uint8_t) (tok + (3 << 6)); \
  |  |  516|   513k|                cf[rc_i] = (tok << 11) | rc; \
  |  |  517|   513k|                rc = rc_i; \
  |  |  518|  9.64M|            } else { \
  |  |  519|  9.64M|                /* 0x1 for tok, 0x7ff as bitmask for rc, 0x41 for level_tok */ \
  |  |  520|  9.64M|                tok *= 0x17ff41; \
  |  |  521|  9.64M|                *level = (uint8_t) tok; \
  |  |  522|  9.64M|                /* tok ? (tok << 11) | rc : 0 */ \
  |  |  523|  9.64M|                tok = (tok >> 9) & (rc + ~0x7ffu); \
  |  |  524|  9.64M|                if (tok) rc = rc_i; \
  |  |  ------------------
  |  |  |  Branch (524:21): [True: 2.60M, False: 7.03M]
  |  |  ------------------
  |  |  525|  9.64M|                cf[rc_i] = tok; \
  |  |  526|  9.64M|            } \
  |  |  527|  10.1M|        } \
  |  |  528|   541k|        /* dc */ \
  |  |  529|   541k|        ctx = (tx_class == TX_CLASS_2D) ? 0 : \
  |  |  ------------------
  |  |  |  Branch (529:15): [Folded, False: 540k]
  |  |  ------------------
  |  |  530|   540k|            get_lo_ctx(levels, tx_class, &mag, lo_ctx_offsets, 0, 0, stride); \
  |  |  531|   540k|        dc_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|   540k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  532|   540k|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (532:13): [Folded, False: 540k]
  |  |  ------------------
  |  |  533|   540k|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n", \
  |  |  534|      0|                   t_dim->ctx, chroma, ctx, dc_tok, ts->msac.rng); \
  |  |  535|   540k|        if (dc_tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (535:13): [True: 61.6k, False: 478k]
  |  |  ------------------
  |  |  536|  61.6k|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (536:17): [Folded, False: 61.6k]
  |  |  ------------------
  |  |  537|  61.6k|                mag = levels[0 * stride + 1] + levels[1 * stride + 0] + \
  |  |  538|      0|                      levels[1 * stride + 1]; \
  |  |  539|  61.6k|            mag &= 63; \
  |  |  540|  61.6k|            ctx = mag > 12 ? 6 : (mag + 1) >> 1; \
  |  |  ------------------
  |  |  |  Branch (540:19): [True: 7.57k, False: 54.1k]
  |  |  ------------------
  |  |  541|  61.6k|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  61.6k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  542|  61.6k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (542:17): [Folded, False: 61.6k]
  |  |  ------------------
  |  |  543|  61.6k|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n", \
  |  |  544|      0|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng); \
  |  |  545|  61.6k|        } \
  |  |  546|   540k|        break
  ------------------
  |  Branch (567:13): [True: 10.1M, False: 774]
  |  Branch (567:13): [True: 10.1M, False: 634]
  ------------------
  568|   541k|        }
  569|   292k|        case TX_CLASS_V: {
  ------------------
  |  Branch (569:9): [True: 292k, False: 13.3M]
  ------------------
  570|   292k|            const uint8_t (*const lo_ctx_offsets)[5] = NULL;
  571|   292k|            const ptrdiff_t stride = 16;
  572|   292k|            const unsigned shift = slw + 2, shift2 = slh + 2;
  573|   292k|            const unsigned mask = (4 << slw) - 1;
  574|   292k|            memset(levels, 0, stride * ((4 << slw) + 2));
  575|   292k|            DECODE_COEFS_CLASS(TX_CLASS_V);
  ------------------
  |  |  459|   292k|        unsigned x, y; \
  |  |  460|   292k|        uint8_t *level; \
  |  |  461|   292k|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (461:13): [Folded, False: 292k]
  |  |  ------------------
  |  |  462|   292k|            rc = scan[eob], x = rc >> shift, y = rc & mask; \
  |  |  463|   292k|        else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (463:18): [Folded, False: 292k]
  |  |  ------------------
  |  |  464|   292k|            /* Transposing reduces the stride and padding requirements */ \
  |  |  465|   292k|            x = eob & mask, y = eob >> shift, rc = eob; \
  |  |  466|   292k|        else /* tx_class == TX_CLASS_V */ \
  |  |  467|   292k|            x = eob & mask, y = eob >> shift, rc = (x << shift2) | y; \
  |  |  468|   292k|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (468:13): [Folded, False: 292k]
  |  |  ------------------
  |  |  469|   292k|            printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  470|      0|                   t_dim->ctx, chroma, ctx, eob, rc, tok, ts->msac.rng); \
  |  |  471|   292k|        if (eob_tok == 2) { \
  |  |  ------------------
  |  |  |  Branch (471:13): [True: 3.43k, False: 288k]
  |  |  ------------------
  |  |  472|  3.43k|            ctx = (tx_class == TX_CLASS_2D ? (x | y) > 1 : y != 0) ? 14 : 7; \
  |  |  ------------------
  |  |  |  Branch (472:19): [True: 3.11k, False: 314]
  |  |  |  Branch (472:20): [Folded, False: 3.43k]
  |  |  ------------------
  |  |  473|  3.43k|            tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  3.43k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  474|  3.43k|            level_tok = tok + (3 << 6); \
  |  |  475|  3.43k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (475:17): [Folded, False: 3.43k]
  |  |  ------------------
  |  |  476|  3.43k|                printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  477|      0|                       imin(t_dim->ctx, 3), chroma, ctx, eob, rc, tok, \
  |  |  478|      0|                       ts->msac.rng); \
  |  |  479|  3.43k|        } \
  |  |  480|   292k|        cf[rc] = tok << 11; \
  |  |  481|   292k|        if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (481:13): [Folded, False: 292k]
  |  |  ------------------
  |  |  482|   292k|            level = levels + rc; \
  |  |  483|   292k|        else \
  |  |  484|   292k|            level = levels + x * stride + y; \
  |  |  485|   292k|        *level = (uint8_t) level_tok; \
  |  |  486|  6.21M|        for (int i = eob - 1; i > 0; i--) { /* ac */ \
  |  |  ------------------
  |  |  |  Branch (486:31): [True: 5.92M, False: 292k]
  |  |  ------------------
  |  |  487|  5.92M|            unsigned rc_i; \
  |  |  488|  5.92M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (488:17): [Folded, False: 5.92M]
  |  |  ------------------
  |  |  489|  5.92M|                rc_i = scan[i], x = rc_i >> shift, y = rc_i & mask; \
  |  |  490|  5.92M|            else if (tx_class == TX_CLASS_H) \
  |  |  ------------------
  |  |  |  Branch (490:22): [Folded, False: 5.92M]
  |  |  ------------------
  |  |  491|  5.92M|                x = i & mask, y = i >> shift, rc_i = i; \
  |  |  492|  5.92M|            else /* tx_class == TX_CLASS_V */ \
  |  |  493|  5.92M|                x = i & mask, y = i >> shift, rc_i = (x << shift2) | y; \
  |  |  494|  5.92M|            assert(x < 32 && y < 32); \
  |  |  495|  5.92M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (495:17): [Folded, False: 5.92M]
  |  |  ------------------
  |  |  496|  5.92M|                level = levels + rc_i; \
  |  |  497|  5.92M|            else \
  |  |  498|  5.92M|                level = levels + x * stride + y; \
  |  |  499|  5.92M|            ctx = get_lo_ctx(level, tx_class, &mag, lo_ctx_offsets, x, y, stride); \
  |  |  500|  5.92M|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (500:17): [Folded, False: 5.92M]
  |  |  ------------------
  |  |  501|  5.92M|                y |= x; \
  |  |  502|  5.92M|            tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|  5.92M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  503|  5.92M|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (503:17): [Folded, False: 5.92M]
  |  |  ------------------
  |  |  504|  5.92M|                printf("Post-lo_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  505|      0|                       t_dim->ctx, chroma, ctx, i, rc_i, tok, ts->msac.rng); \
  |  |  506|  5.92M|            if (tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (506:17): [True: 241k, False: 5.68M]
  |  |  ------------------
  |  |  507|   241k|                mag &= 63; \
  |  |  508|   241k|                ctx = (y > (tx_class == TX_CLASS_2D) ? 14 : 7) + \
  |  |  ------------------
  |  |  |  Branch (508:24): [True: 144k, False: 96.6k]
  |  |  ------------------
  |  |  509|   241k|                      (mag > 12 ? 6 : (mag + 1) >> 1); \
  |  |  ------------------
  |  |  |  Branch (509:24): [True: 22.9k, False: 218k]
  |  |  ------------------
  |  |  510|   241k|                tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|   241k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  511|   241k|                if (dbg) \
  |  |  ------------------
  |  |  |  Branch (511:21): [Folded, False: 241k]
  |  |  ------------------
  |  |  512|   241k|                    printf("Post-hi_tok[%d][%d][%d][%d=%d=%d]: r=%d\n", \
  |  |  513|      0|                           imin(t_dim->ctx, 3), chroma, ctx, i, rc_i, tok, \
  |  |  514|      0|                           ts->msac.rng); \
  |  |  515|   241k|                *level = (uint8_t) (tok + (3 << 6)); \
  |  |  516|   241k|                cf[rc_i] = (tok << 11) | rc; \
  |  |  517|   241k|                rc = rc_i; \
  |  |  518|  5.68M|            } else { \
  |  |  519|  5.68M|                /* 0x1 for tok, 0x7ff as bitmask for rc, 0x41 for level_tok */ \
  |  |  520|  5.68M|                tok *= 0x17ff41; \
  |  |  521|  5.68M|                *level = (uint8_t) tok; \
  |  |  522|  5.68M|                /* tok ? (tok << 11) | rc : 0 */ \
  |  |  523|  5.68M|                tok = (tok >> 9) & (rc + ~0x7ffu); \
  |  |  524|  5.68M|                if (tok) rc = rc_i; \
  |  |  ------------------
  |  |  |  Branch (524:21): [True: 1.42M, False: 4.26M]
  |  |  ------------------
  |  |  525|  5.68M|                cf[rc_i] = tok; \
  |  |  526|  5.68M|            } \
  |  |  527|  5.92M|        } \
  |  |  528|   292k|        /* dc */ \
  |  |  529|   292k|        ctx = (tx_class == TX_CLASS_2D) ? 0 : \
  |  |  ------------------
  |  |  |  Branch (529:15): [Folded, False: 292k]
  |  |  ------------------
  |  |  530|   292k|            get_lo_ctx(levels, tx_class, &mag, lo_ctx_offsets, 0, 0, stride); \
  |  |  531|   292k|        dc_tok = dav1d_msac_decode_symbol_adapt4(&ts->msac, lo_cdf[ctx], 3); \
  |  |  ------------------
  |  |  |  |   47|   292k|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  |  |  ------------------
  |  |  532|   292k|        if (dbg) \
  |  |  ------------------
  |  |  |  Branch (532:13): [Folded, False: 292k]
  |  |  ------------------
  |  |  533|   292k|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n", \
  |  |  534|      0|                   t_dim->ctx, chroma, ctx, dc_tok, ts->msac.rng); \
  |  |  535|   292k|        if (dc_tok == 3) { \
  |  |  ------------------
  |  |  |  Branch (535:13): [True: 26.5k, False: 266k]
  |  |  ------------------
  |  |  536|  26.5k|            if (tx_class == TX_CLASS_2D) \
  |  |  ------------------
  |  |  |  Branch (536:17): [Folded, False: 26.5k]
  |  |  ------------------
  |  |  537|  26.5k|                mag = levels[0 * stride + 1] + levels[1 * stride + 0] + \
  |  |  538|      0|                      levels[1 * stride + 1]; \
  |  |  539|  26.5k|            mag &= 63; \
  |  |  540|  26.5k|            ctx = mag > 12 ? 6 : (mag + 1) >> 1; \
  |  |  ------------------
  |  |  |  Branch (540:19): [True: 3.90k, False: 22.6k]
  |  |  ------------------
  |  |  541|  26.5k|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[ctx]); \
  |  |  ------------------
  |  |  |  |   49|  26.5k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  |  |  ------------------
  |  |  542|  26.5k|            if (dbg) \
  |  |  ------------------
  |  |  |  Branch (542:17): [Folded, False: 26.5k]
  |  |  ------------------
  |  |  543|  26.5k|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n", \
  |  |  544|      0|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng); \
  |  |  545|  26.5k|        } \
  |  |  546|   292k|        break
  ------------------
  |  Branch (575:13): [True: 5.92M, False: 18.4E]
  |  Branch (575:13): [True: 5.92M, False: 113]
  ------------------
  576|   292k|        }
  577|      0|#undef DECODE_COEFS_CLASS
  578|      0|        default: assert(0);
  ------------------
  |  Branch (578:9): [True: 0, False: 13.6M]
  |  Branch (578:18): [Folded, False: 0]
  ------------------
  579|  13.6M|        }
  580|  13.6M|    } else { // dc-only
  581|  4.46M|        int tok_br = dav1d_msac_decode_symbol_adapt4(&ts->msac, eob_cdf[0], 2);
  ------------------
  |  |   47|  4.46M|#define dav1d_msac_decode_symbol_adapt4  dav1d_msac_decode_symbol_adapt4_sse2
  ------------------
  582|  4.46M|        dc_tok = 1 + tok_br;
  583|  4.46M|        if (dbg)
  ------------------
  |  Branch (583:13): [Folded, False: 4.46M]
  ------------------
  584|      0|            printf("Post-dc_lo_tok[%d][%d][%d][%d]: r=%d\n",
  585|      0|                   t_dim->ctx, chroma, 0, dc_tok, ts->msac.rng);
  586|  4.46M|        if (tok_br == 2) {
  ------------------
  |  Branch (586:13): [True: 279k, False: 4.18M]
  ------------------
  587|   279k|            dc_tok = dav1d_msac_decode_hi_tok(&ts->msac, hi_cdf[0]);
  ------------------
  |  |   49|   279k|#define dav1d_msac_decode_hi_tok         dav1d_msac_decode_hi_tok_sse2
  ------------------
  588|   279k|            if (dbg)
  ------------------
  |  Branch (588:17): [Folded, False: 279k]
  ------------------
  589|      0|                printf("Post-dc_hi_tok[%d][%d][0][%d]: r=%d\n",
  590|      0|                       imin(t_dim->ctx, 3), chroma, dc_tok, ts->msac.rng);
  591|   279k|        }
  592|  4.46M|        rc = 0;
  593|  4.46M|    }
  594|       |
  595|       |    // residual and sign
  596|  18.1M|    const uint16_t *const dq_tbl = ts->dq[b->seg_id][plane];
  597|  18.1M|    const uint8_t *const qm_tbl = *txtp < IDTX ? f->qm[tx][plane] : NULL;
  ------------------
  |  Branch (597:35): [True: 9.09M, False: 9.03M]
  ------------------
  598|  18.1M|    const int dq_shift = imax(0, t_dim->ctx - 2);
  599|  18.1M|    const int cf_max = ~(~127U << (BITDEPTH == 8 ? 8 : f->cur.p.bpc));
  ------------------
  |  Branch (599:36): [True: 4.86M, Folded]
  ------------------
  600|  18.1M|    unsigned cul_level, dc_sign_level;
  601|       |
  602|  18.1M|    if (!dc_tok) {
  ------------------
  |  Branch (602:9): [True: 3.20M, False: 14.9M]
  ------------------
  603|  3.20M|        cul_level = 0;
  604|  3.20M|        dc_sign_level = 1 << 6;
  605|  3.20M|        if (qm_tbl) goto ac_qm;
  ------------------
  |  Branch (605:13): [True: 193k, False: 3.00M]
  ------------------
  606|  3.00M|        goto ac_noqm;
  607|  3.20M|    }
  608|       |
  609|  14.9M|    const int dc_sign_ctx = get_dc_sign_ctx(tx, a, l);
  610|  14.9M|    uint16_t *const dc_sign_cdf = ts->cdf.coef.dc_sign[chroma][dc_sign_ctx];
  611|  14.9M|    const int dc_sign = dav1d_msac_decode_bool_adapt(&ts->msac, dc_sign_cdf);
  ------------------
  |  |   52|  14.9M|#define dav1d_msac_decode_bool_adapt     dav1d_msac_decode_bool_adapt_sse2
  ------------------
  612|  14.9M|    if (dbg)
  ------------------
  |  Branch (612:9): [Folded, False: 14.9M]
  ------------------
  613|      0|        printf("Post-dc_sign[%d][%d][%d]: r=%d\n",
  614|      0|               chroma, dc_sign_ctx, dc_sign, ts->msac.rng);
  615|       |
  616|  14.9M|    int dc_dq = dq_tbl[0];
  617|  14.9M|    dc_sign_level = (dc_sign - 1) & (2 << 6);
  618|       |
  619|  14.9M|    if (qm_tbl) {
  ------------------
  |  Branch (619:9): [True: 1.51M, False: 13.4M]
  ------------------
  620|  1.51M|        dc_dq = (dc_dq * qm_tbl[0] + 16) >> 5;
  621|       |
  622|  1.51M|        if (dc_tok == 15) {
  ------------------
  |  Branch (622:13): [True: 25.9k, False: 1.48M]
  ------------------
  623|  25.9k|            dc_tok = read_golomb(&ts->msac) + 15;
  624|  25.9k|            if (dbg)
  ------------------
  |  Branch (624:17): [Folded, False: 25.9k]
  ------------------
  625|      0|                printf("Post-dc_residual[%d->%d]: r=%d\n",
  626|      0|                       dc_tok - 15, dc_tok, ts->msac.rng);
  627|       |
  628|  25.9k|            dc_tok &= 0xfffff;
  629|  25.9k|            dc_dq = (dc_dq * dc_tok) & 0xffffff;
  630|  1.48M|        } else {
  631|  1.48M|            dc_dq *= dc_tok;
  632|  1.48M|            assert(dc_dq <= 0xffffff);
  ------------------
  |  Branch (632:13): [True: 1.48M, False: 18.4E]
  ------------------
  633|  1.48M|        }
  634|  1.51M|        cul_level = dc_tok;
  635|  1.51M|        dc_dq >>= dq_shift;
  636|  1.51M|        dc_dq = umin(dc_dq, cf_max + dc_sign);
  637|  1.51M|        cf[0] = (coef) (dc_sign ? -dc_dq : dc_dq);
  ------------------
  |  Branch (637:25): [True: 727k, False: 787k]
  ------------------
  638|       |
  639|  1.51M|        if (rc) ac_qm: {
  ------------------
  |  Branch (639:13): [True: 619k, False: 895k]
  ------------------
  640|  1.43M|            const unsigned ac_dq = dq_tbl[1];
  641|  10.3M|            do {
  642|  10.3M|                const int sign = dav1d_msac_decode_bool_equi(&ts->msac);
  ------------------
  |  |   53|  10.3M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  643|  10.3M|                if (dbg)
  ------------------
  |  Branch (643:21): [Folded, False: 10.3M]
  ------------------
  644|      0|                    printf("Post-sign[%d=%d]: r=%d\n", rc, sign, ts->msac.rng);
  645|  10.3M|                const unsigned rc_tok = cf[rc];
  646|  10.3M|                unsigned tok, dq = (ac_dq * qm_tbl[rc] + 16) >> 5;
  647|  10.3M|                int dq_sat;
  648|       |
  649|  10.3M|                if (rc_tok >= (15 << 11)) {
  ------------------
  |  Branch (649:21): [True: 530k, False: 9.84M]
  ------------------
  650|   530k|                    tok = read_golomb(&ts->msac) + 15;
  651|   530k|                    if (dbg)
  ------------------
  |  Branch (651:25): [Folded, False: 530k]
  ------------------
  652|      0|                        printf("Post-residual[%d=%d->%d]: r=%d\n",
  653|      0|                               rc, tok - 15, tok, ts->msac.rng);
  654|       |
  655|   530k|                    tok &= 0xfffff;
  656|   530k|                    dq = (dq * tok) & 0xffffff;
  657|  9.84M|                } else {
  658|  9.84M|                    tok = rc_tok >> 11;
  659|  9.84M|                    dq *= tok;
  660|  9.84M|                    assert(dq <= 0xffffff);
  ------------------
  |  Branch (660:21): [True: 9.84M, False: 2]
  ------------------
  661|  9.84M|                }
  662|  10.3M|                cul_level += tok;
  663|  10.3M|                dq >>= dq_shift;
  664|  10.3M|                dq_sat = umin(dq, cf_max + sign);
  665|  10.3M|                cf[rc] = (coef) (sign ? -dq_sat : dq_sat);
  ------------------
  |  Branch (665:34): [True: 5.23M, False: 5.13M]
  ------------------
  666|       |
  667|  10.3M|                rc = rc_tok & 0x3ff;
  668|  10.3M|            } while (rc);
  ------------------
  |  Branch (668:22): [True: 9.55M, False: 812k]
  ------------------
  669|  1.43M|        }
  670|  13.4M|    } else {
  671|       |        // non-qmatrix is the common case and allows for additional optimizations
  672|  13.4M|        if (dc_tok == 15) {
  ------------------
  |  Branch (672:13): [True: 669k, False: 12.7M]
  ------------------
  673|   669k|            dc_tok = read_golomb(&ts->msac) + 15;
  674|   669k|            if (dbg)
  ------------------
  |  Branch (674:17): [Folded, False: 669k]
  ------------------
  675|      0|                printf("Post-dc_residual[%d->%d]: r=%d\n",
  676|      0|                       dc_tok - 15, dc_tok, ts->msac.rng);
  677|       |
  678|   669k|            dc_tok &= 0xfffff;
  679|   669k|            dc_dq = ((dc_dq * dc_tok) & 0xffffff) >> dq_shift;
  680|   669k|            dc_dq = umin(dc_dq, cf_max + dc_sign);
  681|  12.7M|        } else {
  682|  12.7M|            dc_dq = ((dc_dq * dc_tok) >> dq_shift);
  683|  12.7M|            assert(dc_dq <= cf_max);
  ------------------
  |  Branch (683:13): [True: 12.8M, False: 18.4E]
  ------------------
  684|  12.7M|        }
  685|  13.4M|        cul_level = dc_tok;
  686|  13.4M|        cf[0] = (coef) (dc_sign ? -dc_dq : dc_dq);
  ------------------
  |  Branch (686:25): [True: 6.30M, False: 7.17M]
  ------------------
  687|       |
  688|  22.7M|        if (rc) ac_noqm: {
  ------------------
  |  Branch (688:13): [True: 9.88M, False: 3.59M]
  ------------------
  689|  22.7M|            const unsigned ac_dq = dq_tbl[1];
  690|  98.7M|            do {
  691|  98.7M|                const int sign = dav1d_msac_decode_bool_equi(&ts->msac);
  ------------------
  |  |   53|  98.7M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  692|  98.7M|                if (dbg)
  ------------------
  |  Branch (692:21): [Folded, False: 98.7M]
  ------------------
  693|      0|                    printf("Post-sign[%d=%d]: r=%d\n", rc, sign, ts->msac.rng);
  694|  98.7M|                const unsigned rc_tok = cf[rc];
  695|  98.7M|                unsigned tok;
  696|  98.7M|                int dq;
  697|       |
  698|       |                // residual
  699|  98.7M|                if (rc_tok >= (15 << 11)) {
  ------------------
  |  Branch (699:21): [True: 1.85M, False: 96.8M]
  ------------------
  700|  1.85M|                    tok = read_golomb(&ts->msac) + 15;
  701|  1.85M|                    if (dbg)
  ------------------
  |  Branch (701:25): [Folded, False: 1.85M]
  ------------------
  702|      0|                        printf("Post-residual[%d=%d->%d]: r=%d\n",
  703|      0|                               rc, tok - 15, tok, ts->msac.rng);
  704|       |
  705|       |                    // coefficient parsing, see 5.11.39
  706|  1.85M|                    tok &= 0xfffff;
  707|       |
  708|       |                    // dequant, see 7.12.3
  709|  1.85M|                    dq = ((ac_dq * tok) & 0xffffff) >> dq_shift;
  710|  1.85M|                    dq = umin(dq, cf_max + sign);
  711|  96.8M|                } else {
  712|       |                    // cannot exceed cf_max, so we can avoid the clipping
  713|  96.8M|                    tok = rc_tok >> 11;
  714|  96.8M|                    dq = ((ac_dq * tok) >> dq_shift);
  715|  96.8M|                    assert(dq <= cf_max);
  ------------------
  |  Branch (715:21): [True: 96.7M, False: 101k]
  ------------------
  716|  96.8M|                }
  717|  98.6M|                cul_level += tok;
  718|  98.6M|                cf[rc] = (coef) (sign ? -dq : dq);
  ------------------
  |  Branch (718:34): [True: 49.8M, False: 48.7M]
  ------------------
  719|       |
  720|  98.6M|                rc = rc_tok & 0x3ff; // next non-zero rc, zero if eob
  721|  98.6M|            } while (rc);
  ------------------
  |  Branch (721:22): [True: 85.8M, False: 12.7M]
  ------------------
  722|  22.7M|        }
  723|  13.4M|    }
  724|       |
  725|       |    // context
  726|  18.0M|    *res_ctx = umin(cul_level, 63) | dc_sign_level;
  727|       |
  728|  18.0M|    return eob;
  729|  14.9M|}
recon_tmpl.c:get_skip_ctx:
   65|  49.9M|{
   66|  49.9M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
   67|       |
   68|  49.9M|    if (chroma) {
  ------------------
  |  Branch (68:9): [True: 32.0M, False: 17.9M]
  ------------------
   69|  32.0M|        const int ss_ver = layout == DAV1D_PIXEL_LAYOUT_I420;
   70|  32.0M|        const int ss_hor = layout != DAV1D_PIXEL_LAYOUT_I444;
   71|  32.0M|        const int not_one_blk = b_dim[2] - (!!b_dim[2] && ss_hor) > t_dim->lw ||
  ------------------
  |  Branch (71:33): [True: 21.7M, False: 10.2M]
  |  Branch (71:45): [True: 30.9M, False: 1.10M]
  |  Branch (71:59): [True: 5.54M, False: 25.4M]
  ------------------
   72|  10.2M|                                b_dim[3] - (!!b_dim[3] && ss_ver) > t_dim->lh;
  ------------------
  |  Branch (72:33): [True: 689k, False: 9.58M]
  |  Branch (72:45): [True: 8.60M, False: 1.67M]
  |  Branch (72:59): [True: 4.32M, False: 4.27M]
  ------------------
   73|  32.0M|        unsigned ca, cl;
   74|       |
   75|  32.0M|#define MERGE_CTX(dir, type, no_val) \
   76|  32.0M|        c##dir = *(const type *) dir != no_val; \
   77|  32.0M|        break
   78|       |
   79|  32.0M|        switch (t_dim->lw) {
   80|       |        /* For some reason the MSVC CRT _wassert() function is not flagged as
   81|       |         * __declspec(noreturn), so when using those headers the compiler will
   82|       |         * expect execution to continue after an assertion has been triggered
   83|       |         * and will therefore complain about the use of uninitialized variables
   84|       |         * when compiled in debug mode if we put the default case at the end. */
   85|      0|        default: assert(0); /* fall-through */
  ------------------
  |  Branch (85:9): [True: 0, False: 32.0M]
  |  Branch (85:18): [Folded, False: 0]
  ------------------
   86|  21.4M|        case TX_4X4:   MERGE_CTX(a, uint8_t,  0x40);
  ------------------
  |  |   76|  21.4M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  21.4M|        break
  ------------------
  |  Branch (86:9): [True: 21.4M, False: 10.6M]
  ------------------
   87|  3.41M|        case TX_8X8:   MERGE_CTX(a, uint16_t, 0x4040);
  ------------------
  |  |   76|  3.41M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  3.41M|        break
  ------------------
  |  Branch (87:9): [True: 3.41M, False: 28.6M]
  ------------------
   88|  2.49M|        case TX_16X16: MERGE_CTX(a, uint32_t, 0x40404040U);
  ------------------
  |  |   76|  2.49M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  2.49M|        break
  ------------------
  |  Branch (88:9): [True: 2.49M, False: 29.5M]
  ------------------
   89|  4.77M|        case TX_32X32: MERGE_CTX(a, uint64_t, 0x4040404040404040ULL);
  ------------------
  |  |   76|  4.77M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  4.77M|        break
  ------------------
  |  Branch (89:9): [True: 4.77M, False: 27.2M]
  ------------------
   90|  32.0M|        }
   91|  32.0M|        switch (t_dim->lh) {
   92|      0|        default: assert(0); /* fall-through */
  ------------------
  |  Branch (92:9): [True: 0, False: 32.0M]
  |  Branch (92:18): [Folded, False: 0]
  ------------------
   93|  22.5M|        case TX_4X4:   MERGE_CTX(l, uint8_t,  0x40);
  ------------------
  |  |   76|  22.5M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  22.5M|        break
  ------------------
  |  Branch (93:9): [True: 22.5M, False: 9.56M]
  ------------------
   94|  3.07M|        case TX_8X8:   MERGE_CTX(l, uint16_t, 0x4040);
  ------------------
  |  |   76|  3.07M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  3.07M|        break
  ------------------
  |  Branch (94:9): [True: 3.07M, False: 28.9M]
  ------------------
   95|  2.17M|        case TX_16X16: MERGE_CTX(l, uint32_t, 0x40404040U);
  ------------------
  |  |   76|  2.17M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  2.17M|        break
  ------------------
  |  Branch (95:9): [True: 2.17M, False: 29.8M]
  ------------------
   96|  4.38M|        case TX_32X32: MERGE_CTX(l, uint64_t, 0x4040404040404040ULL);
  ------------------
  |  |   76|  4.38M|        c##dir = *(const type *) dir != no_val; \
  |  |   77|  4.38M|        break
  ------------------
  |  Branch (96:9): [True: 4.38M, False: 27.6M]
  ------------------
   97|  32.0M|        }
   98|  32.0M|#undef MERGE_CTX
   99|       |
  100|  32.0M|        return 7 + not_one_blk * 3 + ca + cl;
  101|  32.0M|    } else if (b_dim[2] == t_dim->lw && b_dim[3] == t_dim->lh) {
  ------------------
  |  Branch (101:16): [True: 6.20M, False: 11.7M]
  |  Branch (101:41): [True: 5.98M, False: 224k]
  ------------------
  102|  5.98M|        return 0;
  103|  11.9M|    } else {
  104|  11.9M|        unsigned la, ll;
  105|       |
  106|  11.9M|#define MERGE_CTX(dir, type, tx) \
  107|  11.9M|        if (tx == TX_64X64) { \
  108|  11.9M|            uint64_t tmp = *(const uint64_t *) dir; \
  109|  11.9M|            tmp |= *(const uint64_t *) &dir[8]; \
  110|  11.9M|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  111|  11.9M|        } else \
  112|  11.9M|            l##dir = *(const type *) dir; \
  113|  11.9M|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  114|  11.9M|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  115|  11.9M|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  116|  11.9M|        break
  117|       |
  118|  11.9M|        switch (t_dim->lw) {
  119|      0|        default: assert(0); /* fall-through */
  ------------------
  |  Branch (119:9): [True: 0, False: 11.9M]
  |  Branch (119:18): [Folded, False: 0]
  ------------------
  120|  10.4M|        case TX_4X4:   MERGE_CTX(a, uint8_t,  TX_4X4);
  ------------------
  |  |  107|  10.4M|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 10.4M]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|  10.4M|            l##dir = *(const type *) dir; \
  |  |  113|  10.4M|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 10.4M]
  |  |  ------------------
  |  |  114|  10.4M|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [Folded, False: 10.4M]
  |  |  ------------------
  |  |  115|  10.4M|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [Folded, False: 10.4M]
  |  |  ------------------
  |  |  116|  10.4M|        break
  ------------------
  |  Branch (120:9): [True: 10.4M, False: 1.50M]
  ------------------
  121|   688k|        case TX_8X8:   MERGE_CTX(a, uint16_t, TX_8X8);
  ------------------
  |  |  107|   688k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 688k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|   688k|            l##dir = *(const type *) dir; \
  |  |  113|   688k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 688k]
  |  |  ------------------
  |  |  114|   688k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [Folded, False: 688k]
  |  |  ------------------
  |  |  115|   688k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 688k, Folded]
  |  |  ------------------
  |  |  116|   688k|        break
  ------------------
  |  Branch (121:9): [True: 688k, False: 11.2M]
  ------------------
  122|   372k|        case TX_16X16: MERGE_CTX(a, uint32_t, TX_16X16);
  ------------------
  |  |  107|   372k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 372k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|   372k|            l##dir = *(const type *) dir; \
  |  |  113|   372k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 372k]
  |  |  ------------------
  |  |  114|   372k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 372k, Folded]
  |  |  ------------------
  |  |  115|   372k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 372k, Folded]
  |  |  ------------------
  |  |  116|   372k|        break
  ------------------
  |  Branch (122:9): [True: 372k, False: 11.5M]
  ------------------
  123|  41.1k|        case TX_32X32: MERGE_CTX(a, uint32_t, TX_32X32);
  ------------------
  |  |  107|  41.1k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 41.1k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|  41.1k|            l##dir = *(const type *) dir; \
  |  |  113|  41.1k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [True: 41.1k, Folded]
  |  |  ------------------
  |  |  114|  41.1k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 41.1k, Folded]
  |  |  ------------------
  |  |  115|  41.1k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 41.1k, Folded]
  |  |  ------------------
  |  |  116|  41.1k|        break
  ------------------
  |  Branch (123:9): [True: 41.1k, False: 11.8M]
  ------------------
  124|   709k|        case TX_64X64: MERGE_CTX(a, uint32_t, TX_64X64);
  ------------------
  |  |  107|   709k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [True: 709k, Folded]
  |  |  ------------------
  |  |  108|   709k|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|   709k|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|   709k|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|   709k|        } else \
  |  |  112|  18.4E|            l##dir = *(const type *) dir; \
  |  |  113|   709k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 709k]
  |  |  ------------------
  |  |  114|   709k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 709k, Folded]
  |  |  ------------------
  |  |  115|   709k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 709k, Folded]
  |  |  ------------------
  |  |  116|   709k|        break
  ------------------
  |  Branch (124:9): [True: 709k, False: 11.2M]
  ------------------
  125|  11.9M|        }
  126|  12.2M|        switch (t_dim->lh) {
  127|      0|        default: assert(0); /* fall-through */
  ------------------
  |  Branch (127:9): [True: 0, False: 12.2M]
  |  Branch (127:18): [Folded, False: 0]
  ------------------
  128|  10.4M|        case TX_4X4:   MERGE_CTX(l, uint8_t,  TX_4X4);
  ------------------
  |  |  107|  10.4M|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 10.4M]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|  10.4M|            l##dir = *(const type *) dir; \
  |  |  113|  10.4M|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 10.4M]
  |  |  ------------------
  |  |  114|  10.4M|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [Folded, False: 10.4M]
  |  |  ------------------
  |  |  115|  10.4M|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [Folded, False: 10.4M]
  |  |  ------------------
  |  |  116|  10.4M|        break
  ------------------
  |  Branch (128:9): [True: 10.4M, False: 1.79M]
  ------------------
  129|   679k|        case TX_8X8:   MERGE_CTX(l, uint16_t, TX_8X8);
  ------------------
  |  |  107|   679k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 679k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|   679k|            l##dir = *(const type *) dir; \
  |  |  113|   679k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 679k]
  |  |  ------------------
  |  |  114|   679k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [Folded, False: 679k]
  |  |  ------------------
  |  |  115|   679k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 679k, Folded]
  |  |  ------------------
  |  |  116|   679k|        break
  ------------------
  |  Branch (129:9): [True: 679k, False: 11.5M]
  ------------------
  130|   361k|        case TX_16X16: MERGE_CTX(l, uint32_t, TX_16X16);
  ------------------
  |  |  107|   361k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 361k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|   361k|            l##dir = *(const type *) dir; \
  |  |  113|   361k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 361k]
  |  |  ------------------
  |  |  114|   361k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 361k, Folded]
  |  |  ------------------
  |  |  115|   361k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 361k, Folded]
  |  |  ------------------
  |  |  116|   361k|        break
  ------------------
  |  Branch (130:9): [True: 361k, False: 11.8M]
  ------------------
  131|  41.1k|        case TX_32X32: MERGE_CTX(l, uint32_t, TX_32X32);
  ------------------
  |  |  107|  41.1k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [Folded, False: 41.1k]
  |  |  ------------------
  |  |  108|      0|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|      0|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|      0|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|      0|        } else \
  |  |  112|  41.1k|            l##dir = *(const type *) dir; \
  |  |  113|  41.1k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [True: 41.1k, Folded]
  |  |  ------------------
  |  |  114|  41.1k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 41.1k, Folded]
  |  |  ------------------
  |  |  115|  41.1k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 41.1k, Folded]
  |  |  ------------------
  |  |  116|  41.1k|        break
  ------------------
  |  Branch (131:9): [True: 41.1k, False: 12.1M]
  ------------------
  132|   709k|        case TX_64X64: MERGE_CTX(l, uint32_t, TX_64X64);
  ------------------
  |  |  107|   709k|        if (tx == TX_64X64) { \
  |  |  ------------------
  |  |  |  Branch (107:13): [True: 709k, Folded]
  |  |  ------------------
  |  |  108|   709k|            uint64_t tmp = *(const uint64_t *) dir; \
  |  |  109|   709k|            tmp |= *(const uint64_t *) &dir[8]; \
  |  |  110|   709k|            l##dir = (unsigned) (tmp >> 32) | (unsigned) tmp; \
  |  |  111|   709k|        } else \
  |  |  112|   709k|            l##dir = *(const type *) dir; \
  |  |  113|   709k|        if (tx == TX_32X32) l##dir |= *(const type *) &dir[sizeof(type)]; \
  |  |  ------------------
  |  |  |  Branch (113:13): [Folded, False: 709k]
  |  |  ------------------
  |  |  114|   709k|        if (tx >= TX_16X16) l##dir |= l##dir >> 16; \
  |  |  ------------------
  |  |  |  Branch (114:13): [True: 709k, Folded]
  |  |  ------------------
  |  |  115|   709k|        if (tx >= TX_8X8)   l##dir |= l##dir >> 8; \
  |  |  ------------------
  |  |  |  Branch (115:13): [True: 709k, Folded]
  |  |  ------------------
  |  |  116|   709k|        break
  ------------------
  |  Branch (132:9): [True: 709k, False: 11.5M]
  ------------------
  133|  12.2M|        }
  134|  12.2M|#undef MERGE_CTX
  135|       |
  136|  12.2M|        return dav1d_skip_ctx[umin(la & 0x3F, 4)][umin(ll & 0x3F, 4)];
  137|  12.2M|    }
  138|  49.9M|}
recon_tmpl.c:get_lo_ctx:
  304|   219M|{
  305|   219M|    unsigned mag = levels[0 * stride + 1] + levels[1 * stride + 0];
  306|   219M|    unsigned offset;
  307|   219M|    if (tx_class == TX_CLASS_2D) {
  ------------------
  |  Branch (307:9): [True: 207M, False: 12.2M]
  ------------------
  308|   207M|        mag += levels[1 * stride + 1];
  309|   207M|        *hi_mag = mag;
  310|   207M|        mag += levels[0 * stride + 2] + levels[2 * stride + 0];
  311|   207M|        offset = ctx_offsets[umin(y, 4)][umin(x, 4)];
  312|   207M|    } else {
  313|  12.2M|        mag += levels[0 * stride + 2];
  314|  12.2M|        *hi_mag = mag;
  315|  12.2M|        mag += levels[0 * stride + 3] + levels[0 * stride + 4];
  316|  12.2M|        offset = 26 + (y > 1 ? 10 : y * 5);
  ------------------
  |  Branch (316:24): [True: 8.00M, False: 4.29M]
  ------------------
  317|  12.2M|    }
  318|   219M|    return offset + (mag > 512 ? 4 : (mag + 64) >> 7);
  ------------------
  |  Branch (318:22): [True: 12.7M, False: 207M]
  ------------------
  319|   219M|}
recon_tmpl.c:get_dc_sign_ctx:
  143|  14.9M|{
  144|  14.9M|    uint64_t mask = 0xC0C0C0C0C0C0C0C0ULL, mul = 0x0101010101010101ULL;
  145|  14.9M|    int s;
  146|       |
  147|  14.9M|#if ARCH_X86_64 && defined(__GNUC__)
  148|       |    /* Coerce compilers into producing better code. For some reason
  149|       |     * every x86-64 compiler is awful at handling 64-bit constants. */
  150|  14.9M|    __asm__("" : "+r"(mask), "+r"(mul));
  151|  14.9M|#endif
  152|       |
  153|  14.9M|    switch(tx) {
  154|      0|    default: assert(0); /* fall-through */
  ------------------
  |  Branch (154:5): [True: 0, False: 14.9M]
  |  Branch (154:14): [Folded, False: 0]
  ------------------
  155|  7.49M|    case TX_4X4: {
  ------------------
  |  Branch (155:5): [True: 7.49M, False: 7.47M]
  ------------------
  156|  7.49M|        int t = *(const uint8_t *) a >> 6;
  157|  7.49M|        t    += *(const uint8_t *) l >> 6;
  158|  7.49M|        s = t - 1 - 1;
  159|  7.49M|        break;
  160|      0|    }
  161|  1.23M|    case TX_8X8: {
  ------------------
  |  Branch (161:5): [True: 1.23M, False: 13.7M]
  ------------------
  162|  1.23M|        uint32_t t = *(const uint16_t *) a & (uint32_t) mask;
  163|  1.23M|        t         += *(const uint16_t *) l & (uint32_t) mask;
  164|  1.23M|        t *= 0x04040404U;
  165|  1.23M|        s = (int) (t >> 24) - 2 - 2;
  166|  1.23M|        break;
  167|      0|    }
  168|   821k|    case TX_16X16: {
  ------------------
  |  Branch (168:5): [True: 821k, False: 14.1M]
  ------------------
  169|   821k|        uint32_t t = (*(const uint32_t *) a & (uint32_t) mask) >> 6;
  170|   821k|        t         += (*(const uint32_t *) l & (uint32_t) mask) >> 6;
  171|   821k|        t *= (uint32_t) mul;
  172|   821k|        s = (int) (t >> 24) - 4 - 4;
  173|   821k|        break;
  174|      0|    }
  175|  1.03M|    case TX_32X32: {
  ------------------
  |  Branch (175:5): [True: 1.03M, False: 13.9M]
  ------------------
  176|  1.03M|        uint64_t t = (*(const uint64_t *) a & mask) >> 6;
  177|  1.03M|        t         += (*(const uint64_t *) l & mask) >> 6;
  178|  1.03M|        t *= mul;
  179|  1.03M|        s = (int) (t >> 56) - 8 - 8;
  180|  1.03M|        break;
  181|      0|    }
  182|   596k|    case TX_64X64: {
  ------------------
  |  Branch (182:5): [True: 596k, False: 14.3M]
  ------------------
  183|   596k|        uint64_t t = (*(const uint64_t *) &a[0] & mask) >> 6;
  184|   596k|        t         += (*(const uint64_t *) &a[8] & mask) >> 6;
  185|   596k|        t         += (*(const uint64_t *) &l[0] & mask) >> 6;
  186|   596k|        t         += (*(const uint64_t *) &l[8] & mask) >> 6;
  187|   596k|        t *= mul;
  188|   596k|        s = (int) (t >> 56) - 16 - 16;
  189|   596k|        break;
  190|      0|    }
  191|   380k|    case RTX_4X8: {
  ------------------
  |  Branch (191:5): [True: 380k, False: 14.5M]
  ------------------
  192|   380k|        uint32_t t = *(const uint8_t  *) a & (uint32_t) mask;
  193|   380k|        t         += *(const uint16_t *) l & (uint32_t) mask;
  194|   380k|        t *= 0x04040404U;
  195|   380k|        s = (int) (t >> 24) - 1 - 2;
  196|   380k|        break;
  197|      0|    }
  198|   645k|    case RTX_8X4: {
  ------------------
  |  Branch (198:5): [True: 645k, False: 14.3M]
  ------------------
  199|   645k|        uint32_t t = *(const uint16_t *) a & (uint32_t) mask;
  200|   645k|        t         += *(const uint8_t  *) l & (uint32_t) mask;
  201|   645k|        t *= 0x04040404U;
  202|   645k|        s = (int) (t >> 24) - 2 - 1;
  203|   645k|        break;
  204|      0|    }
  205|   331k|    case RTX_8X16: {
  ------------------
  |  Branch (205:5): [True: 331k, False: 14.6M]
  ------------------
  206|   331k|        uint32_t t = *(const uint16_t *) a & (uint32_t) mask;
  207|   331k|        t         += *(const uint32_t *) l & (uint32_t) mask;
  208|   331k|        t = (t >> 6) * (uint32_t) mul;
  209|   331k|        s = (int) (t >> 24) - 2 - 4;
  210|   331k|        break;
  211|      0|    }
  212|   672k|    case RTX_16X8: {
  ------------------
  |  Branch (212:5): [True: 672k, False: 14.3M]
  ------------------
  213|   672k|        uint32_t t = *(const uint32_t *) a & (uint32_t) mask;
  214|   672k|        t         += *(const uint16_t *) l & (uint32_t) mask;
  215|   672k|        t = (t >> 6) * (uint32_t) mul;
  216|   672k|        s = (int) (t >> 24) - 4 - 2;
  217|   672k|        break;
  218|      0|    }
  219|   161k|    case RTX_16X32: {
  ------------------
  |  Branch (219:5): [True: 161k, False: 14.8M]
  ------------------
  220|   161k|        uint64_t t = *(const uint32_t *) a & (uint32_t) mask;
  221|   161k|        t         += *(const uint64_t *) l & mask;
  222|   161k|        t = (t >> 6) * mul;
  223|   161k|        s = (int) (t >> 56) - 4 - 8;
  224|   161k|        break;
  225|      0|    }
  226|   224k|    case RTX_32X16: {
  ------------------
  |  Branch (226:5): [True: 224k, False: 14.7M]
  ------------------
  227|   224k|        uint64_t t = *(const uint64_t *) a & mask;
  228|   224k|        t         += *(const uint32_t *) l & (uint32_t) mask;
  229|   224k|        t = (t >> 6) * mul;
  230|   224k|        s = (int) (t >> 56) - 8 - 4;
  231|   224k|        break;
  232|      0|    }
  233|  30.2k|    case RTX_32X64: {
  ------------------
  |  Branch (233:5): [True: 30.2k, False: 14.9M]
  ------------------
  234|  30.2k|        uint64_t t = (*(const uint64_t *) &a[0] & mask) >> 6;
  235|  30.2k|        t         += (*(const uint64_t *) &l[0] & mask) >> 6;
  236|  30.2k|        t         += (*(const uint64_t *) &l[8] & mask) >> 6;
  237|  30.2k|        t *= mul;
  238|  30.2k|        s = (int) (t >> 56) - 8 - 16;
  239|  30.2k|        break;
  240|      0|    }
  241|  48.7k|    case RTX_64X32: {
  ------------------
  |  Branch (241:5): [True: 48.7k, False: 14.9M]
  ------------------
  242|  48.7k|        uint64_t t = (*(const uint64_t *) &a[0] & mask) >> 6;
  243|  48.7k|        t         += (*(const uint64_t *) &a[8] & mask) >> 6;
  244|  48.7k|        t         += (*(const uint64_t *) &l[0] & mask) >> 6;
  245|  48.7k|        t *= mul;
  246|  48.7k|        s = (int) (t >> 56) - 16 - 8;
  247|  48.7k|        break;
  248|      0|    }
  249|   237k|    case RTX_4X16: {
  ------------------
  |  Branch (249:5): [True: 237k, False: 14.7M]
  ------------------
  250|   237k|        uint32_t t = *(const uint8_t  *) a & (uint32_t) mask;
  251|   237k|        t         += *(const uint32_t *) l & (uint32_t) mask;
  252|   237k|        t = (t >> 6) * (uint32_t) mul;
  253|   237k|        s = (int) (t >> 24) - 1 - 4;
  254|   237k|        break;
  255|      0|    }
  256|   511k|    case RTX_16X4: {
  ------------------
  |  Branch (256:5): [True: 511k, False: 14.4M]
  ------------------
  257|   511k|        uint32_t t = *(const uint32_t *) a & (uint32_t) mask;
  258|   511k|        t         += *(const uint8_t  *) l & (uint32_t) mask;
  259|   511k|        t = (t >> 6) * (uint32_t) mul;
  260|   511k|        s = (int) (t >> 24) - 4 - 1;
  261|   511k|        break;
  262|      0|    }
  263|   168k|    case RTX_8X32: {
  ------------------
  |  Branch (263:5): [True: 168k, False: 14.8M]
  ------------------
  264|   168k|        uint64_t t = *(const uint16_t *) a & (uint32_t) mask;
  265|   168k|        t         += *(const uint64_t *) l & mask;
  266|   168k|        t = (t >> 6) * mul;
  267|   168k|        s = (int) (t >> 56) - 2 - 8;
  268|   168k|        break;
  269|      0|    }
  270|   301k|    case RTX_32X8: {
  ------------------
  |  Branch (270:5): [True: 301k, False: 14.6M]
  ------------------
  271|   301k|        uint64_t t = *(const uint64_t *) a & mask;
  272|   301k|        t         += *(const uint16_t *) l & (uint32_t) mask;
  273|   301k|        t = (t >> 6) * mul;
  274|   301k|        s = (int) (t >> 56) - 8 - 2;
  275|   301k|        break;
  276|      0|    }
  277|  53.1k|    case RTX_16X64: {
  ------------------
  |  Branch (277:5): [True: 53.1k, False: 14.9M]
  ------------------
  278|  53.1k|        uint64_t t = *(const uint32_t *) a & (uint32_t) mask;
  279|  53.1k|        t         += *(const uint64_t *) &l[0] & mask;
  280|  53.1k|        t = (t >> 6) + ((*(const uint64_t *) &l[8] & mask) >> 6);
  281|  53.1k|        t *= mul;
  282|  53.1k|        s = (int) (t >> 56) - 4 - 16;
  283|  53.1k|        break;
  284|      0|    }
  285|  72.0k|    case RTX_64X16: {
  ------------------
  |  Branch (285:5): [True: 72.0k, False: 14.9M]
  ------------------
  286|  72.0k|        uint64_t t = *(const uint64_t *) &a[0] & mask;
  287|  72.0k|        t         += *(const uint32_t *) l & (uint32_t) mask;
  288|  72.0k|        t = (t >> 6) + ((*(const uint64_t *) &a[8] & mask) >> 6);
  289|  72.0k|        t *= mul;
  290|  72.0k|        s = (int) (t >> 56) - 16 - 4;
  291|  72.0k|        break;
  292|      0|    }
  293|  14.9M|    }
  294|       |
  295|  14.9M|    return (s != 0) + (s > 0);
  296|  14.9M|}
recon_tmpl.c:read_golomb:
   49|  3.08M|static inline unsigned read_golomb(MsacContext *const msac) {
   50|  3.08M|    int len = 0;
   51|  3.08M|    unsigned val = 1;
   52|       |
   53|  7.01M|    while (!dav1d_msac_decode_bool_equi(msac) && len < 32) len++;
  ------------------
  |  |   53|  7.01M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (53:12): [True: 3.93M, False: 3.08M]
  |  Branch (53:50): [True: 3.93M, False: 18.4E]
  ------------------
   54|  7.01M|    while (len--) val = (val << 1) + dav1d_msac_decode_bool_equi(msac);
  ------------------
  |  |   53|  3.93M|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (54:12): [True: 3.93M, False: 3.08M]
  ------------------
   55|       |
   56|  3.08M|    return val - 1;
   57|  3.08M|}
recon_tmpl.c:mc:
  944|  5.12M|{
  945|  5.12M|    assert((dst8 != NULL) ^ (dst16 != NULL));
  ------------------
  |  Branch (945:5): [True: 5.12M, False: 181]
  ------------------
  946|  5.12M|    const Dav1dFrameContext *const f = t->f;
  947|  5.12M|    const int ss_ver = !!pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (947:24): [True: 2.62M, False: 2.49M]
  |  Branch (947:32): [True: 2.13M, False: 487k]
  ------------------
  948|  5.12M|    const int ss_hor = !!pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (948:24): [True: 2.62M, False: 2.49M]
  |  Branch (948:32): [True: 2.14M, False: 484k]
  ------------------
  949|  5.12M|    const int h_mul = 4 >> ss_hor, v_mul = 4 >> ss_ver;
  950|  5.12M|    const int mvx = mv.x, mvy = mv.y;
  951|  5.12M|    const int mx = mvx & (15 >> !ss_hor), my = mvy & (15 >> !ss_ver);
  952|  5.12M|    ptrdiff_t ref_stride = refp->p.stride[!!pl];
  953|  5.12M|    const pixel *ref;
  954|       |
  955|  5.12M|    if (refp->p.p.w == f->cur.p.w && refp->p.p.h == f->cur.p.h) {
  ------------------
  |  Branch (955:9): [True: 4.59M, False: 532k]
  |  Branch (955:38): [True: 4.45M, False: 137k]
  ------------------
  956|  4.45M|        const int dx = bx * h_mul + (mvx >> (3 + ss_hor));
  957|  4.45M|        const int dy = by * v_mul + (mvy >> (3 + ss_ver));
  958|  4.45M|        int w, h;
  959|       |
  960|  4.45M|        if (refp->p.data[0] != f->cur.data[0]) { // i.e. not for intrabc
  ------------------
  |  Branch (960:13): [True: 4.40M, False: 53.9k]
  ------------------
  961|  4.40M|            w = (f->cur.p.w + ss_hor) >> ss_hor;
  962|  4.40M|            h = (f->cur.p.h + ss_ver) >> ss_ver;
  963|  4.40M|        } else {
  964|  53.9k|            w = f->bw * 4 >> ss_hor;
  965|  53.9k|            h = f->bh * 4 >> ss_ver;
  966|  53.9k|        }
  967|  4.45M|        if (dx < !!mx * 3 || dy < !!my * 3 ||
  ------------------
  |  Branch (967:13): [True: 97.7k, False: 4.35M]
  |  Branch (967:30): [True: 55.8k, False: 4.30M]
  ------------------
  968|  4.30M|            dx + bw4 * h_mul + !!mx * 4 > w ||
  ------------------
  |  Branch (968:13): [True: 917k, False: 3.38M]
  ------------------
  969|  3.38M|            dy + bh4 * v_mul + !!my * 4 > h)
  ------------------
  |  Branch (969:13): [True: 185k, False: 3.19M]
  ------------------
  970|  1.26M|        {
  971|  1.26M|            pixel *const emu_edge_buf = bitfn(t->scratch.emu_edge);
  ------------------
  |  |   51|  1.26M|#define bitfn(x) x##_8bpc
  ------------------
  972|  1.26M|            f->dsp->mc.emu_edge(bw4 * h_mul + !!mx * 7, bh4 * v_mul + !!my * 7,
  973|  1.26M|                                w, h, dx - !!mx * 3, dy - !!my * 3,
  974|  1.26M|                                emu_edge_buf, 192 * sizeof(pixel),
  975|  1.26M|                                refp->p.data[pl], ref_stride);
  976|  1.26M|            ref = &emu_edge_buf[192 * !!my * 3 + !!mx * 3];
  977|  1.26M|            ref_stride = 192 * sizeof(pixel);
  978|  3.19M|        } else {
  979|  3.19M|            ref = ((pixel *) refp->p.data[pl]) + PXSTRIDE(ref_stride) * dy + dx;
  ------------------
  |  |   53|  3.19M|#define PXSTRIDE(x) (x)
  ------------------
  980|  3.19M|        }
  981|       |
  982|  4.45M|        if (dst8 != NULL) {
  ------------------
  |  Branch (982:13): [True: 3.89M, False: 564k]
  ------------------
  983|  3.89M|            f->dsp->mc.mc[filter_2d](dst8, dst_stride, ref, ref_stride, bw4 * h_mul,
  984|  3.89M|                                     bh4 * v_mul, mx << !ss_hor, my << !ss_ver
  985|  3.89M|                                     HIGHBD_CALL_SUFFIX);
  986|  3.89M|        } else {
  987|   564k|            f->dsp->mc.mct[filter_2d](dst16, ref, ref_stride, bw4 * h_mul,
  988|   564k|                                      bh4 * v_mul, mx << !ss_hor, my << !ss_ver
  989|   564k|                                      HIGHBD_CALL_SUFFIX);
  990|   564k|        }
  991|  4.45M|    } else {
  992|   670k|        assert(refp != &f->sr_cur);
  ------------------
  |  Branch (992:9): [True: 673k, False: 18.4E]
  ------------------
  993|       |
  994|   673k|        const int orig_pos_y = (by * v_mul << 4) + mvy * (1 << !ss_ver);
  995|   673k|        const int orig_pos_x = (bx * h_mul << 4) + mvx * (1 << !ss_hor);
  996|   673k|#define scale_mv(res, val, scale) do { \
  997|   673k|            const int64_t tmp = (int64_t)(val) * scale + (scale - 0x4000) * 8; \
  998|   673k|            res = apply_sign64((int) ((llabs(tmp) + 128) >> 8), tmp) + 32;     \
  999|   673k|        } while (0)
 1000|   673k|        int pos_y, pos_x;
 1001|   673k|        scale_mv(pos_x, orig_pos_x, f->svc[refidx][0].scale);
  ------------------
  |  |  996|   673k|#define scale_mv(res, val, scale) do { \
  |  |  997|   673k|            const int64_t tmp = (int64_t)(val) * scale + (scale - 0x4000) * 8; \
  |  |  998|   673k|            res = apply_sign64((int) ((llabs(tmp) + 128) >> 8), tmp) + 32;     \
  |  |  999|   673k|        } while (0)
  |  |  ------------------
  |  |  |  Branch (999:18): [Folded, False: 673k]
  |  |  ------------------
  ------------------
 1002|   673k|        scale_mv(pos_y, orig_pos_y, f->svc[refidx][1].scale);
  ------------------
  |  |  996|   673k|#define scale_mv(res, val, scale) do { \
  |  |  997|   673k|            const int64_t tmp = (int64_t)(val) * scale + (scale - 0x4000) * 8; \
  |  |  998|   673k|            res = apply_sign64((int) ((llabs(tmp) + 128) >> 8), tmp) + 32;     \
  |  |  999|   673k|        } while (0)
  |  |  ------------------
  |  |  |  Branch (999:18): [Folded, False: 673k]
  |  |  ------------------
  ------------------
 1003|   673k|#undef scale_mv
 1004|   673k|        const int left = pos_x >> 10;
 1005|   673k|        const int top = pos_y >> 10;
 1006|   673k|        const int right =
 1007|   673k|            ((pos_x + (bw4 * h_mul - 1) * f->svc[refidx][0].step) >> 10) + 1;
 1008|   673k|        const int bottom =
 1009|   673k|            ((pos_y + (bh4 * v_mul - 1) * f->svc[refidx][1].step) >> 10) + 1;
 1010|       |
 1011|   673k|        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|   673k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 673k]
  |  |  ------------------
  |  |   35|   673k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   673k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1012|      0|            printf("Off %dx%d [%d,%d,%d], size %dx%d [%d,%d]\n",
 1013|      0|                   left, top, orig_pos_x, f->svc[refidx][0].scale, refidx,
 1014|      0|                   right-left, bottom-top,
 1015|      0|                   f->svc[refidx][0].step, f->svc[refidx][1].step);
 1016|       |
 1017|   673k|        const int w = (refp->p.p.w + ss_hor) >> ss_hor;
 1018|   673k|        const int h = (refp->p.p.h + ss_ver) >> ss_ver;
 1019|   673k|        if (left < 3 || top < 3 || right + 4 > w || bottom + 4 > h) {
  ------------------
  |  Branch (1019:13): [True: 132k, False: 541k]
  |  Branch (1019:25): [True: 35.9k, False: 505k]
  |  Branch (1019:36): [True: 110k, False: 394k]
  |  Branch (1019:53): [True: 51.9k, False: 343k]
  ------------------
 1020|   327k|            pixel *const emu_edge_buf = bitfn(t->scratch.emu_edge);
  ------------------
  |  |   51|   327k|#define bitfn(x) x##_8bpc
  ------------------
 1021|   327k|            f->dsp->mc.emu_edge(right - left + 7, bottom - top + 7,
 1022|   327k|                                w, h, left - 3, top - 3,
 1023|   327k|                                emu_edge_buf, 320 * sizeof(pixel),
 1024|   327k|                                refp->p.data[pl], ref_stride);
 1025|   327k|            ref = &emu_edge_buf[320 * 3 + 3];
 1026|   327k|            ref_stride = 320 * sizeof(pixel);
 1027|   327k|            if (DEBUG_BLOCK_INFO) printf("Emu\n");
  ------------------
  |  |   34|   327k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 327k]
  |  |  ------------------
  |  |   35|   327k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   327k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1028|   345k|        } else {
 1029|   345k|            ref = ((pixel *) refp->p.data[pl]) + PXSTRIDE(ref_stride) * top + left;
  ------------------
  |  |   53|   345k|#define PXSTRIDE(x) (x)
  ------------------
 1030|   345k|        }
 1031|       |
 1032|   673k|        if (dst8 != NULL) {
  ------------------
  |  Branch (1032:13): [True: 569k, False: 104k]
  ------------------
 1033|   569k|            f->dsp->mc.mc_scaled[filter_2d](dst8, dst_stride, ref, ref_stride,
 1034|   569k|                                            bw4 * h_mul, bh4 * v_mul,
 1035|   569k|                                            pos_x & 0x3ff, pos_y & 0x3ff,
 1036|   569k|                                            f->svc[refidx][0].step,
 1037|   569k|                                            f->svc[refidx][1].step
 1038|   569k|                                            HIGHBD_CALL_SUFFIX);
 1039|   569k|        } else {
 1040|   104k|            f->dsp->mc.mct_scaled[filter_2d](dst16, ref, ref_stride,
 1041|   104k|                                             bw4 * h_mul, bh4 * v_mul,
 1042|   104k|                                             pos_x & 0x3ff, pos_y & 0x3ff,
 1043|   104k|                                             f->svc[refidx][0].step,
 1044|   104k|                                             f->svc[refidx][1].step
 1045|   104k|                                             HIGHBD_CALL_SUFFIX);
 1046|   104k|        }
 1047|   673k|    }
 1048|       |
 1049|  5.12M|    return 0;
 1050|  5.12M|}
recon_tmpl.c:warp_affine:
 1120|  1.03M|{
 1121|  1.03M|    assert((dst8 != NULL) ^ (dst16 != NULL));
  ------------------
  |  Branch (1121:5): [True: 1.03M, False: 18.4E]
  ------------------
 1122|  1.03M|    const Dav1dFrameContext *const f = t->f;
 1123|  1.03M|    const Dav1dDSPContext *const dsp = f->dsp;
 1124|  1.03M|    const int ss_ver = !!pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (1124:24): [True: 162k, False: 874k]
  |  Branch (1124:32): [True: 134k, False: 28.1k]
  ------------------
 1125|  1.03M|    const int ss_hor = !!pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (1125:24): [True: 162k, False: 874k]
  |  Branch (1125:32): [True: 134k, False: 28.0k]
  ------------------
 1126|  1.03M|    const int h_mul = 4 >> ss_hor, v_mul = 4 >> ss_ver;
 1127|  1.03M|    assert(!((b_dim[0] * h_mul) & 7) && !((b_dim[1] * v_mul) & 7));
  ------------------
  |  Branch (1127:5): [True: 1.03M, False: 18.4E]
  |  Branch (1127:5): [True: 1.03M, False: 7]
  ------------------
 1128|  1.03M|    const int32_t *const mat = wmp->matrix;
 1129|  1.03M|    const int width = (refp->p.p.w + ss_hor) >> ss_hor;
 1130|  1.03M|    const int height = (refp->p.p.h + ss_ver) >> ss_ver;
 1131|       |
 1132|  9.01M|    for (int y = 0; y < b_dim[1] * v_mul; y += 8) {
  ------------------
  |  Branch (1132:21): [True: 7.97M, False: 1.03M]
  ------------------
 1133|  7.97M|        const int src_y = t->by * 4 + ((y + 4) << ss_ver);
 1134|  7.97M|        const int64_t mat3_y = (int64_t) mat[3] * src_y + mat[0];
 1135|  7.97M|        const int64_t mat5_y = (int64_t) mat[5] * src_y + mat[1];
 1136|  80.6M|        for (int x = 0; x < b_dim[0] * h_mul; x += 8) {
  ------------------
  |  Branch (1136:25): [True: 72.7M, False: 7.97M]
  ------------------
 1137|       |            // calculate transformation relative to center of 8x8 block in
 1138|       |            // luma pixel units
 1139|  72.7M|            const int src_x = t->bx * 4 + ((x + 4) << ss_hor);
 1140|  72.7M|            const int64_t mvx = ((int64_t) mat[2] * src_x + mat3_y) >> ss_hor;
 1141|  72.7M|            const int64_t mvy = ((int64_t) mat[4] * src_x + mat5_y) >> ss_ver;
 1142|       |
 1143|  72.7M|            const int dx = (int) (mvx >> 16) - 4;
 1144|  72.7M|            const int mx = (((int) mvx & 0xffff) - wmp->u.p.alpha * 4 -
 1145|  72.7M|                                                   wmp->u.p.beta  * 7) & ~0x3f;
 1146|  72.7M|            const int dy = (int) (mvy >> 16) - 4;
 1147|  72.7M|            const int my = (((int) mvy & 0xffff) - wmp->u.p.gamma * 4 -
 1148|  72.7M|                                                   wmp->u.p.delta * 4) & ~0x3f;
 1149|       |
 1150|  72.7M|            const pixel *ref_ptr;
 1151|  72.7M|            ptrdiff_t ref_stride = refp->p.stride[!!pl];
 1152|       |
 1153|  72.7M|            if (dx < 3 || dx + 8 + 4 > width || dy < 3 || dy + 8 + 4 > height) {
  ------------------
  |  Branch (1153:17): [True: 4.33M, False: 68.3M]
  |  Branch (1153:27): [True: 17.9M, False: 50.4M]
  |  Branch (1153:49): [True: 82.3k, False: 50.3M]
  |  Branch (1153:59): [True: 167k, False: 50.2M]
  ------------------
 1154|  24.2M|                pixel *const emu_edge_buf = bitfn(t->scratch.emu_edge);
  ------------------
  |  |   51|  24.2M|#define bitfn(x) x##_8bpc
  ------------------
 1155|  24.2M|                f->dsp->mc.emu_edge(15, 15, width, height, dx - 3, dy - 3,
 1156|  24.2M|                                    emu_edge_buf, 32 * sizeof(pixel),
 1157|  24.2M|                                    refp->p.data[pl], ref_stride);
 1158|  24.2M|                ref_ptr = &emu_edge_buf[32 * 3 + 3];
 1159|  24.2M|                ref_stride = 32 * sizeof(pixel);
 1160|  48.4M|            } else {
 1161|  48.4M|                ref_ptr = ((pixel *) refp->p.data[pl]) + PXSTRIDE(ref_stride) * dy + dx;
  ------------------
  |  |   53|  48.4M|#define PXSTRIDE(x) (x)
  ------------------
 1162|  48.4M|            }
 1163|  72.7M|            if (dst16 != NULL)
  ------------------
  |  Branch (1163:17): [True: 106k, False: 72.6M]
  ------------------
 1164|   106k|                dsp->mc.warp8x8t(&dst16[x], dstride, ref_ptr, ref_stride,
 1165|   106k|                                 wmp->u.abcd, mx, my HIGHBD_CALL_SUFFIX);
 1166|  72.6M|            else
 1167|  72.6M|                dsp->mc.warp8x8(&dst8[x], dstride, ref_ptr, ref_stride,
 1168|  72.6M|                                wmp->u.abcd, mx, my HIGHBD_CALL_SUFFIX);
 1169|  72.7M|        }
 1170|  7.97M|        if (dst8) dst8  += 8 * PXSTRIDE(dstride);
  ------------------
  |  |   53|  7.94M|#define PXSTRIDE(x) (x)
  ------------------
  |  Branch (1170:13): [True: 7.94M, False: 38.3k]
  ------------------
 1171|  38.3k|        else      dst16 += 8 * dstride;
 1172|  7.97M|    }
 1173|  1.03M|    return 0;
 1174|  1.03M|}
recon_tmpl.c:obmc:
 1056|   575k|{
 1057|   575k|    assert(!(t->bx & 1) && !(t->by & 1));
  ------------------
  |  Branch (1057:5): [True: 575k, False: 18.4E]
  |  Branch (1057:5): [True: 575k, False: 18.4E]
  ------------------
 1058|   575k|    const Dav1dFrameContext *const f = t->f;
 1059|   575k|    /*const*/ refmvs_block **r = &t->rt.r[(t->by & 31) + 5];
 1060|   575k|    pixel *const lap = bitfn(t->scratch.lap);
  ------------------
  |  |   51|   575k|#define bitfn(x) x##_8bpc
  ------------------
 1061|   575k|    const int ss_ver = !!pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (1061:24): [True: 364k, False: 210k]
  |  Branch (1061:32): [True: 288k, False: 76.5k]
  ------------------
 1062|   575k|    const int ss_hor = !!pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (1062:24): [True: 364k, False: 210k]
  |  Branch (1062:32): [True: 289k, False: 75.9k]
  ------------------
 1063|   575k|    const int h_mul = 4 >> ss_hor, v_mul = 4 >> ss_ver;
 1064|   575k|    int res;
 1065|       |
 1066|   575k|    if (t->by > t->ts->tiling.row_start &&
  ------------------
  |  Branch (1066:9): [True: 558k, False: 17.2k]
  ------------------
 1067|   558k|        (!pl || b_dim[0] * h_mul + b_dim[1] * v_mul >= 16))
  ------------------
  |  Branch (1067:10): [True: 204k, False: 354k]
  |  Branch (1067:17): [True: 151k, False: 202k]
  ------------------
 1068|   356k|    {
 1069|   759k|        for (int i = 0, x = 0; x < w4 && i < imin(b_dim[2], 4); ) {
  ------------------
  |  Branch (1069:32): [True: 404k, False: 355k]
  |  Branch (1069:42): [True: 403k, False: 1.21k]
  ------------------
 1070|       |            // only odd blocks are considered for overlap handling, hence +1
 1071|   403k|            const refmvs_block *const a_r = &r[-1][t->bx + x + 1];
 1072|   403k|            const uint8_t *const a_b_dim = dav1d_block_dimensions[a_r->bs];
 1073|   403k|            const int step4 = iclip(a_b_dim[0], 2, 16);
 1074|       |
 1075|   403k|            if (a_r->ref.ref[0] > 0) {
  ------------------
  |  Branch (1075:17): [True: 379k, False: 23.8k]
  ------------------
 1076|   379k|                const int ow4 = imin(step4, b_dim[0]);
 1077|   379k|                const int oh4 = imin(b_dim[1], 16) >> 1;
 1078|   379k|                res = mc(t, lap, NULL, ow4 * h_mul * sizeof(pixel), ow4, (oh4 * 3 + 3) >> 2,
 1079|   379k|                         t->bx + x, t->by, pl, a_r->mv.mv[0],
 1080|   379k|                         &f->refp[a_r->ref.ref[0] - 1], a_r->ref.ref[0] - 1,
 1081|   379k|                         dav1d_filter_2d[t->a->filter[1][bx4 + x + 1]][t->a->filter[0][bx4 + x + 1]]);
 1082|   379k|                if (res) return res;
  ------------------
  |  Branch (1082:21): [True: 0, False: 379k]
  ------------------
 1083|   379k|                f->dsp->mc.blend_h(&dst[x * h_mul], dst_stride, lap,
 1084|   379k|                                   h_mul * ow4, v_mul * oh4);
 1085|   379k|                i++;
 1086|   379k|            }
 1087|   403k|            x += step4;
 1088|   403k|        }
 1089|   356k|    }
 1090|       |
 1091|   575k|    if (t->bx > t->ts->tiling.col_start)
  ------------------
  |  Branch (1091:9): [True: 553k, False: 22.4k]
  ------------------
 1092|  1.15M|        for (int i = 0, y = 0; y < h4 && i < imin(b_dim[3], 4); ) {
  ------------------
  |  Branch (1092:32): [True: 604k, False: 552k]
  |  Branch (1092:42): [True: 603k, False: 1.33k]
  ------------------
 1093|       |            // only odd blocks are considered for overlap handling, hence +1
 1094|   603k|            const refmvs_block *const l_r = &r[y + 1][t->bx - 1];
 1095|   603k|            const uint8_t *const l_b_dim = dav1d_block_dimensions[l_r->bs];
 1096|   603k|            const int step4 = iclip(l_b_dim[1], 2, 16);
 1097|       |
 1098|   603k|            if (l_r->ref.ref[0] > 0) {
  ------------------
  |  Branch (1098:17): [True: 559k, False: 43.4k]
  ------------------
 1099|   559k|                const int ow4 = imin(b_dim[0], 16) >> 1;
 1100|   559k|                const int oh4 = imin(step4, b_dim[1]);
 1101|   559k|                res = mc(t, lap, NULL, h_mul * ow4 * sizeof(pixel), ow4, oh4,
 1102|   559k|                         t->bx, t->by + y, pl, l_r->mv.mv[0],
 1103|   559k|                         &f->refp[l_r->ref.ref[0] - 1], l_r->ref.ref[0] - 1,
 1104|   559k|                         dav1d_filter_2d[t->l.filter[1][by4 + y + 1]][t->l.filter[0][by4 + y + 1]]);
 1105|   559k|                if (res) return res;
  ------------------
  |  Branch (1105:21): [True: 0, False: 559k]
  ------------------
 1106|   559k|                f->dsp->mc.blend_v(&dst[y * v_mul * PXSTRIDE(dst_stride)],
  ------------------
  |  |   53|   559k|#define PXSTRIDE(x) (x)
  ------------------
 1107|   559k|                                   dst_stride, lap, h_mul * ow4, v_mul * oh4);
 1108|   559k|                i++;
 1109|   559k|            }
 1110|   603k|            y += step4;
 1111|   603k|        }
 1112|   575k|    return 0;
 1113|   575k|}
dav1d_read_coef_blocks_16bpc:
  826|  6.05M|{
  827|  6.05M|    const Dav1dFrameContext *const f = t->f;
  828|  6.05M|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  829|  6.05M|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  830|  6.05M|    const int bx4 = t->bx & 31, by4 = t->by & 31;
  831|  6.05M|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
  832|  6.05M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  833|  6.05M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
  834|  6.05M|    const int cbw4 = (bw4 + ss_hor) >> ss_hor, cbh4 = (bh4 + ss_ver) >> ss_ver;
  835|  6.05M|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (835:28): [True: 4.41M, False: 1.64M]
  ------------------
  836|  4.41M|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (836:29): [True: 4.08M, False: 326k]
  |  Branch (836:45): [True: 163k, False: 163k]
  ------------------
  837|  4.24M|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (837:29): [True: 3.89M, False: 352k]
  |  Branch (837:45): [True: 176k, False: 176k]
  ------------------
  838|       |
  839|  6.05M|    if (b->skip) {
  ------------------
  |  Branch (839:9): [True: 2.24M, False: 3.80M]
  ------------------
  840|  2.24M|        BlockContext *const a = t->a;
  841|  2.24M|        dav1d_memset_pow2[b_dim[2]](&a->lcoef[bx4], 0x40);
  842|  2.24M|        dav1d_memset_pow2[b_dim[3]](&t->l.lcoef[by4], 0x40);
  843|  2.24M|        if (has_chroma) {
  ------------------
  |  Branch (843:13): [True: 1.16M, False: 1.08M]
  ------------------
  844|  1.16M|            dav1d_memset_pow2_fn memset_cw = dav1d_memset_pow2[ulog2(cbw4)];
  845|  1.16M|            dav1d_memset_pow2_fn memset_ch = dav1d_memset_pow2[ulog2(cbh4)];
  846|  1.16M|            memset_cw(&a->ccoef[0][cbx4], 0x40);
  847|  1.16M|            memset_cw(&a->ccoef[1][cbx4], 0x40);
  848|  1.16M|            memset_ch(&t->l.ccoef[0][cby4], 0x40);
  849|  1.16M|            memset_ch(&t->l.ccoef[1][cby4], 0x40);
  850|  1.16M|        }
  851|  2.24M|        return;
  852|  2.24M|    }
  853|       |
  854|  3.80M|    Dav1dTileState *const ts = t->ts;
  855|  3.80M|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
  856|  3.80M|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
  857|  3.80M|    assert(t->frame_thread.pass == 1);
  ------------------
  |  Branch (857:5): [True: 3.80M, False: 2.81k]
  ------------------
  858|  3.80M|    assert(!b->skip);
  ------------------
  |  Branch (858:5): [True: 3.80M, False: 18.4E]
  ------------------
  859|  3.80M|    const TxfmInfo *const uv_t_dim = &dav1d_txfm_dimensions[b->uvtx];
  860|  3.80M|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[b->intra ? b->tx : b->max_ytx];
  ------------------
  |  Branch (860:58): [True: 2.63M, False: 1.16M]
  ------------------
  861|  3.80M|    const uint16_t tx_split[2] = { b->tx_split0, b->tx_split1 };
  862|       |
  863|  7.75M|    for (int init_y = 0; init_y < h4; init_y += 16) {
  ------------------
  |  Branch (863:26): [True: 3.94M, False: 3.80M]
  ------------------
  864|  3.94M|        const int sub_h4 = imin(h4, 16 + init_y);
  865|  8.17M|        for (int init_x = 0; init_x < w4; init_x += 16) {
  ------------------
  |  Branch (865:30): [True: 4.23M, False: 3.94M]
  ------------------
  866|  4.23M|            const int sub_w4 = imin(w4, init_x + 16);
  867|  4.23M|            int y_off = !!init_y, y, x;
  868|  9.15M|            for (y = init_y, t->by += init_y; y < sub_h4;
  ------------------
  |  Branch (868:47): [True: 4.92M, False: 4.23M]
  ------------------
  869|  4.92M|                 y += t_dim->h, t->by += t_dim->h, y_off++)
  870|  4.92M|            {
  871|  4.92M|                int x_off = !!init_x;
  872|  19.0M|                for (x = init_x, t->bx += init_x; x < sub_w4;
  ------------------
  |  Branch (872:51): [True: 14.0M, False: 4.92M]
  ------------------
  873|  14.0M|                     x += t_dim->w, t->bx += t_dim->w, x_off++)
  874|  14.0M|                {
  875|  14.0M|                    if (!b->intra) {
  ------------------
  |  Branch (875:25): [True: 5.17M, False: 8.89M]
  ------------------
  876|  5.17M|                        read_coef_tree(t, bs, b, b->max_ytx, 0, tx_split,
  877|  5.17M|                                       x_off, y_off, NULL);
  878|  8.89M|                    } else {
  879|  8.89M|                        uint8_t cf_ctx = 0x40;
  880|  8.89M|                        enum TxfmType txtp;
  881|  8.89M|                        const int eob =
  882|  8.89M|                            decode_coefs(t, &t->a->lcoef[bx4 + x],
  883|  8.89M|                                         &t->l.lcoef[by4 + y], b->tx, bs, b, 1,
  884|  8.89M|                                         0, ts->frame_thread[1].cf, &txtp, &cf_ctx);
  885|  8.89M|                        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  8.89M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 8.89M]
  |  |  ------------------
  |  |   35|  8.89M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  8.89M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  886|      0|                            printf("Post-y-cf-blk[tx=%d,txtp=%d,eob=%d]: r=%d\n",
  887|      0|                                   b->tx, txtp, eob, ts->msac.rng);
  888|  8.89M|                        *ts->frame_thread[1].cbi++ = eob * (1 << 5) + txtp;
  889|  8.89M|                        ts->frame_thread[1].cf += imin(t_dim->w, 8) * imin(t_dim->h, 8) * 16;
  890|  8.89M|                        dav1d_memset_likely_pow2(&t->a->lcoef[bx4 + x], cf_ctx, imin(t_dim->w, f->bw - t->bx));
  891|  8.89M|                        dav1d_memset_likely_pow2(&t->l.lcoef[by4 + y], cf_ctx, imin(t_dim->h, f->bh - t->by));
  892|  8.89M|                    }
  893|  14.0M|                }
  894|  4.92M|                t->bx -= x;
  895|  4.92M|            }
  896|  4.23M|            t->by -= y;
  897|       |
  898|  4.23M|            if (!has_chroma) continue;
  ------------------
  |  Branch (898:17): [True: 927k, False: 3.30M]
  ------------------
  899|       |
  900|  3.30M|            const int sub_ch4 = imin(ch4, (init_y + 16) >> ss_ver);
  901|  3.30M|            const int sub_cw4 = imin(cw4, (init_x + 16) >> ss_hor);
  902|  9.88M|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (902:30): [True: 6.57M, False: 3.30M]
  ------------------
  903|  14.7M|                for (y = init_y >> ss_ver, t->by += init_y; y < sub_ch4;
  ------------------
  |  Branch (903:61): [True: 8.15M, False: 6.57M]
  ------------------
  904|  8.15M|                     y += uv_t_dim->h, t->by += uv_t_dim->h << ss_ver)
  905|  8.15M|                {
  906|  33.8M|                    for (x = init_x >> ss_hor, t->bx += init_x; x < sub_cw4;
  ------------------
  |  Branch (906:65): [True: 25.6M, False: 8.15M]
  ------------------
  907|  25.6M|                         x += uv_t_dim->w, t->bx += uv_t_dim->w << ss_hor)
  908|  25.6M|                    {
  909|  25.6M|                        uint8_t cf_ctx = 0x40;
  910|  25.6M|                        enum TxfmType txtp;
  911|  25.6M|                        if (!b->intra)
  ------------------
  |  Branch (911:29): [True: 9.67M, False: 15.9M]
  ------------------
  912|  9.67M|                            txtp = t->scratch.txtp_map[(by4 + (y << ss_ver)) * 32 +
  913|  9.67M|                                                        bx4 + (x << ss_hor)];
  914|  25.6M|                        const int eob =
  915|  25.6M|                            decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
  916|  25.6M|                                         &t->l.ccoef[pl][cby4 + y], b->uvtx, bs,
  917|  25.6M|                                         b, b->intra, 1 + pl, ts->frame_thread[1].cf,
  918|  25.6M|                                         &txtp, &cf_ctx);
  919|  25.6M|                        if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  25.6M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 25.6M]
  |  |  ------------------
  |  |   35|  25.6M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  25.6M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
  920|      0|                            printf("Post-uv-cf-blk[pl=%d,tx=%d,"
  921|      0|                                   "txtp=%d,eob=%d]: r=%d\n",
  922|      0|                                   pl, b->uvtx, txtp, eob, ts->msac.rng);
  923|  25.6M|                        *ts->frame_thread[1].cbi++ = eob * (1 << 5) + txtp;
  924|  25.6M|                        ts->frame_thread[1].cf += uv_t_dim->w * uv_t_dim->h * 16;
  925|  25.6M|                        int ctw = imin(uv_t_dim->w, (f->bw - t->bx + ss_hor) >> ss_hor);
  926|  25.6M|                        int cth = imin(uv_t_dim->h, (f->bh - t->by + ss_ver) >> ss_ver);
  927|  25.6M|                        dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
  928|  25.6M|                        dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
  929|  25.6M|                    }
  930|  8.15M|                    t->bx -= x << ss_hor;
  931|  8.15M|                }
  932|  6.57M|                t->by -= y << ss_ver;
  933|  6.57M|            }
  934|  3.30M|        }
  935|  3.94M|    }
  936|  3.80M|}
dav1d_recon_b_intra_16bpc:
 1179|   887k|{
 1180|   887k|    Dav1dTileState *const ts = t->ts;
 1181|   887k|    const Dav1dFrameContext *const f = t->f;
 1182|   887k|    const Dav1dDSPContext *const dsp = f->dsp;
 1183|   887k|    const int bx4 = t->bx & 31, by4 = t->by & 31;
 1184|   887k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1185|   887k|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 1186|   887k|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
 1187|   887k|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
 1188|   887k|    const int bw4 = b_dim[0], bh4 = b_dim[1];
 1189|   887k|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
 1190|   887k|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
 1191|   887k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (1191:28): [True: 539k, False: 347k]
  ------------------
 1192|   539k|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (1192:29): [True: 468k, False: 71.7k]
  |  Branch (1192:45): [True: 36.3k, False: 35.4k]
  ------------------
 1193|   504k|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (1193:29): [True: 432k, False: 71.7k]
  |  Branch (1193:45): [True: 36.4k, False: 35.3k]
  ------------------
 1194|   887k|    const TxfmInfo *const t_dim = &dav1d_txfm_dimensions[b->tx];
 1195|   887k|    const TxfmInfo *const uv_t_dim = &dav1d_txfm_dimensions[b->uvtx];
 1196|       |
 1197|       |    // coefficient coding
 1198|   887k|    pixel *const edge = bitfn(t->scratch.edge) + 128;
  ------------------
  |  |   77|   887k|#define bitfn(x) x##_16bpc
  ------------------
 1199|   887k|    const int cbw4 = (bw4 + ss_hor) >> ss_hor, cbh4 = (bh4 + ss_ver) >> ss_ver;
 1200|       |
 1201|   887k|    const int intra_edge_filter_flag = f->seq_hdr->intra_edge_filter << 10;
 1202|       |
 1203|  1.92M|    for (int init_y = 0; init_y < h4; init_y += 16) {
  ------------------
  |  Branch (1203:26): [True: 1.03M, False: 889k]
  ------------------
 1204|  1.03M|        const int sub_h4 = imin(h4, 16 + init_y);
 1205|  1.03M|        const int sub_ch4 = imin(ch4, (init_y + 16) >> ss_ver);
 1206|  2.33M|        for (int init_x = 0; init_x < w4; init_x += 16) {
  ------------------
  |  Branch (1206:30): [True: 1.30M, False: 1.03M]
  ------------------
 1207|  1.30M|            if (b->pal_sz[0]) {
  ------------------
  |  Branch (1207:17): [True: 8.21k, False: 1.29M]
  ------------------
 1208|  8.21k|                pixel *dst = ((pixel *) f->cur.data[0]) +
 1209|  8.21k|                             4 * (t->by * PXSTRIDE(f->cur.stride[0]) + t->bx);
 1210|  8.21k|                const uint8_t *pal_idx;
 1211|  8.21k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (1211:21): [True: 8.21k, False: 0]
  ------------------
 1212|  8.21k|                    const int p = t->frame_thread.pass & 1;
 1213|  8.21k|                    assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1213:21): [True: 8.21k, False: 0]
  ------------------
 1214|  8.21k|                    pal_idx = ts->frame_thread[p].pal_idx;
 1215|  8.21k|                    ts->frame_thread[p].pal_idx += bw4 * bh4 * 8;
 1216|  8.21k|                } else {
 1217|      0|                    pal_idx = t->scratch.pal_idx_y;
 1218|      0|                }
 1219|  8.21k|                const pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (1219:42): [True: 8.21k, False: 0]
  ------------------
 1220|  8.21k|                    f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 1221|  8.21k|                                        ((t->bx >> 1) + (t->by & 1))][0] :
 1222|  8.21k|                    bytefn(t->scratch.pal)[0];
  ------------------
  |  |   87|      0|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  8.21k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1223|  8.21k|                f->dsp->ipred.pal_pred(dst, f->cur.stride[0], pal,
 1224|  8.21k|                                       pal_idx, bw4 * 4, bh4 * 4);
 1225|  8.21k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|  8.21k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 8.21k]
  |  |  ------------------
  |  |   35|  8.21k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  8.21k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1226|      0|                    hex_dump(dst, PXSTRIDE(f->cur.stride[0]),
 1227|      0|                             bw4 * 4, bh4 * 4, "y-pal-pred");
 1228|  8.21k|            }
 1229|       |
 1230|  1.30M|            const int intra_flags = (sm_flag(t->a, bx4) |
 1231|  1.30M|                                     sm_flag(&t->l, by4) |
 1232|  1.30M|                                     intra_edge_filter_flag);
 1233|  1.30M|            const int sb_has_tr = init_x + 16 < w4 ? 1 : init_y ? 0 :
  ------------------
  |  Branch (1233:35): [True: 276k, False: 1.03M]
  |  Branch (1233:58): [True: 143k, False: 887k]
  ------------------
 1234|  1.03M|                              intra_edge_flags & EDGE_I444_TOP_HAS_RIGHT;
 1235|  1.30M|            const int sb_has_bl = init_x ? 0 : init_y + 16 < h4 ? 1 :
  ------------------
  |  Branch (1235:35): [True: 276k, False: 1.03M]
  |  Branch (1235:48): [True: 143k, False: 887k]
  ------------------
 1236|  1.03M|                              intra_edge_flags & EDGE_I444_LEFT_HAS_BOTTOM;
 1237|  1.30M|            int y, x;
 1238|  1.30M|            const int sub_w4 = imin(w4, init_x + 16);
 1239|  3.00M|            for (y = init_y, t->by += init_y; y < sub_h4;
  ------------------
  |  Branch (1239:47): [True: 1.69M, False: 1.30M]
  ------------------
 1240|  1.69M|                 y += t_dim->h, t->by += t_dim->h)
 1241|  1.69M|            {
 1242|  1.69M|                pixel *dst = ((pixel *) f->cur.data[0]) +
 1243|  1.69M|                               4 * (t->by * PXSTRIDE(f->cur.stride[0]) +
 1244|  1.69M|                                    t->bx + init_x);
 1245|  8.02M|                for (x = init_x, t->bx += init_x; x < sub_w4;
  ------------------
  |  Branch (1245:51): [True: 6.32M, False: 1.69M]
  ------------------
 1246|  6.32M|                     x += t_dim->w, t->bx += t_dim->w)
 1247|  6.32M|                {
 1248|  6.32M|                    if (b->pal_sz[0]) goto skip_y_pred;
  ------------------
  |  Branch (1248:25): [True: 62.9k, False: 6.25M]
  ------------------
 1249|       |
 1250|  6.25M|                    int angle = b->y_angle;
 1251|  6.25M|                    const enum EdgeFlags edge_flags =
 1252|  6.25M|                        (((y > init_y || !sb_has_tr) && (x + t_dim->w >= sub_w4)) ?
  ------------------
  |  Branch (1252:28): [True: 4.56M, False: 1.69M]
  |  Branch (1252:42): [True: 471k, False: 1.22M]
  |  Branch (1252:57): [True: 737k, False: 4.30M]
  ------------------
 1253|  5.52M|                             0 : EDGE_I444_TOP_HAS_RIGHT) |
 1254|  6.25M|                        ((x > init_x || (!sb_has_bl && y + t_dim->h >= sub_h4)) ?
  ------------------
  |  Branch (1254:27): [True: 4.58M, False: 1.67M]
  |  Branch (1254:42): [True: 1.06M, False: 611k]
  |  Branch (1254:56): [True: 781k, False: 285k]
  ------------------
 1255|  5.36M|                             0 : EDGE_I444_LEFT_HAS_BOTTOM);
 1256|  6.25M|                    const pixel *top_sb_edge = NULL;
 1257|  6.25M|                    if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1257:25): [True: 750k, False: 5.50M]
  ------------------
 1258|   750k|                        top_sb_edge = f->ipred_edge[0];
 1259|   750k|                        const int sby = t->by >> f->sb_shift;
 1260|   750k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1261|   750k|                    }
 1262|  6.25M|                    const enum IntraPredMode m =
 1263|  6.25M|                        bytefn(dav1d_prepare_intra_edges)(t->bx,
  ------------------
  |  |   87|  6.25M|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  6.25M|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1264|  6.25M|                                                          t->bx > ts->tiling.col_start,
 1265|  6.25M|                                                          t->by,
 1266|  6.25M|                                                          t->by > ts->tiling.row_start,
 1267|  6.25M|                                                          ts->tiling.col_end,
 1268|  6.25M|                                                          ts->tiling.row_end,
 1269|  6.25M|                                                          edge_flags, dst,
 1270|  6.25M|                                                          f->cur.stride[0], top_sb_edge,
 1271|  6.25M|                                                          b->y_mode, &angle,
 1272|  6.25M|                                                          t_dim->w, t_dim->h,
 1273|  6.25M|                                                          f->seq_hdr->intra_edge_filter,
 1274|  6.25M|                                                          edge HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  6.25M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1275|  6.25M|                    dsp->ipred.intra_pred[m](dst, f->cur.stride[0], edge,
 1276|  6.25M|                                             t_dim->w * 4, t_dim->h * 4,
 1277|  6.25M|                                             angle | intra_flags,
 1278|  6.25M|                                             4 * f->bw - 4 * t->bx,
 1279|  6.25M|                                             4 * f->bh - 4 * t->by
 1280|  6.25M|                                             HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  6.25M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1281|       |
 1282|  6.25M|                    if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  6.25M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 6.25M]
  |  |  ------------------
  |  |   35|  6.25M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  6.25M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                  if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1283|      0|                        hex_dump(edge - t_dim->h * 4, t_dim->h * 4,
 1284|      0|                                 t_dim->h * 4, 2, "l");
 1285|      0|                        hex_dump(edge, 0, 1, 1, "tl");
 1286|      0|                        hex_dump(edge + 1, t_dim->w * 4,
 1287|      0|                                 t_dim->w * 4, 2, "t");
 1288|      0|                        hex_dump(dst, f->cur.stride[0],
 1289|      0|                                 t_dim->w * 4, t_dim->h * 4, "y-intra-pred");
 1290|      0|                    }
 1291|       |
 1292|  6.32M|                skip_y_pred: {}
 1293|  6.32M|                    if (!b->skip) {
  ------------------
  |  Branch (1293:25): [True: 853k, False: 5.47M]
  ------------------
 1294|   853k|                        coef *cf;
 1295|   853k|                        int eob;
 1296|   853k|                        enum TxfmType txtp;
 1297|   853k|                        if (t->frame_thread.pass) {
  ------------------
  |  Branch (1297:29): [True: 853k, False: 18.4E]
  ------------------
 1298|   853k|                            const int p = t->frame_thread.pass & 1;
 1299|   853k|                            const int cbi = *ts->frame_thread[p].cbi++;
 1300|   853k|                            cf = ts->frame_thread[p].cf;
 1301|   853k|                            ts->frame_thread[p].cf += imin(t_dim->w, 8) * imin(t_dim->h, 8) * 16;
 1302|   853k|                            eob  = cbi >> 5;
 1303|   853k|                            txtp = cbi & 0x1f;
 1304|  18.4E|                        } else {
 1305|  18.4E|                            uint8_t cf_ctx;
 1306|  18.4E|                            cf = bitfn(t->cf);
  ------------------
  |  |   77|  18.4E|#define bitfn(x) x##_16bpc
  ------------------
 1307|  18.4E|                            eob = decode_coefs(t, &t->a->lcoef[bx4 + x],
 1308|  18.4E|                                               &t->l.lcoef[by4 + y], b->tx, bs,
 1309|  18.4E|                                               b, 1, 0, cf, &txtp, &cf_ctx);
 1310|  18.4E|                            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  18.4E|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 18.4E]
  |  |  ------------------
  |  |   35|  18.4E|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  18.4E|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1311|      0|                                printf("Post-y-cf-blk[tx=%d,txtp=%d,eob=%d]: r=%d\n",
 1312|      0|                                       b->tx, txtp, eob, ts->msac.rng);
 1313|  18.4E|                            dav1d_memset_likely_pow2(&t->a->lcoef[bx4 + x], cf_ctx, imin(t_dim->w, f->bw - t->bx));
 1314|  18.4E|                            dav1d_memset_likely_pow2(&t->l.lcoef[by4 + y], cf_ctx, imin(t_dim->h, f->bh - t->by));
 1315|  18.4E|                        }
 1316|   853k|                        if (eob >= 0) {
  ------------------
  |  Branch (1316:29): [True: 653k, False: 199k]
  ------------------
 1317|   653k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   653k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 653k]
  |  |  ------------------
  |  |   35|   653k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   653k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1318|      0|                                coef_dump(cf, imin(t_dim->h, 8) * 4,
 1319|      0|                                          imin(t_dim->w, 8) * 4, 3, "dq");
 1320|   653k|                            dsp->itx.itxfm_add[b->tx]
 1321|   653k|                                              [txtp](dst,
 1322|   653k|                                                     f->cur.stride[0],
 1323|   653k|                                                     cf, eob HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   653k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1324|   653k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   653k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 653k]
  |  |  ------------------
  |  |   35|   653k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   653k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1325|      0|                                hex_dump(dst, f->cur.stride[0],
 1326|      0|                                         t_dim->w * 4, t_dim->h * 4, "recon");
 1327|   653k|                        }
 1328|  5.47M|                    } else if (!t->frame_thread.pass) {
  ------------------
  |  Branch (1328:32): [True: 0, False: 5.47M]
  ------------------
 1329|      0|                        dav1d_memset_pow2[t_dim->lw](&t->a->lcoef[bx4 + x], 0x40);
 1330|      0|                        dav1d_memset_pow2[t_dim->lh](&t->l.lcoef[by4 + y], 0x40);
 1331|      0|                    }
 1332|  6.32M|                    dst += 4 * t_dim->w;
 1333|  6.32M|                }
 1334|  1.69M|                t->bx -= x;
 1335|  1.69M|            }
 1336|  1.30M|            t->by -= y;
 1337|       |
 1338|  1.30M|            if (!has_chroma) continue;
  ------------------
  |  Branch (1338:17): [True: 641k, False: 666k]
  ------------------
 1339|       |
 1340|   666k|            const ptrdiff_t stride = f->cur.stride[1];
 1341|       |
 1342|   666k|            if (b->uv_mode == CFL_PRED) {
  ------------------
  |  Branch (1342:17): [True: 37.5k, False: 628k]
  ------------------
 1343|  37.5k|                assert(!init_x && !init_y);
  ------------------
  |  Branch (1343:17): [True: 37.5k, False: 0]
  |  Branch (1343:17): [True: 37.5k, False: 0]
  ------------------
 1344|       |
 1345|  37.5k|                int16_t *const ac = t->scratch.ac;
 1346|  37.5k|                pixel *y_src = ((pixel *) f->cur.data[0]) + 4 * (t->bx & ~ss_hor) +
 1347|  37.5k|                                 4 * (t->by & ~ss_ver) * PXSTRIDE(f->cur.stride[0]);
 1348|  37.5k|                const ptrdiff_t uv_off = 4 * ((t->bx >> ss_hor) +
 1349|  37.5k|                                              (t->by >> ss_ver) * PXSTRIDE(stride));
 1350|  37.5k|                pixel *const uv_dst[2] = { ((pixel *) f->cur.data[1]) + uv_off,
 1351|  37.5k|                                           ((pixel *) f->cur.data[2]) + uv_off };
 1352|       |
 1353|  37.5k|                const int furthest_r =
 1354|  37.5k|                    ((cw4 << ss_hor) + t_dim->w - 1) & ~(t_dim->w - 1);
 1355|  37.5k|                const int furthest_b =
 1356|  37.5k|                    ((ch4 << ss_ver) + t_dim->h - 1) & ~(t_dim->h - 1);
 1357|  37.5k|                dsp->ipred.cfl_ac[f->cur.p.layout - 1](ac, y_src, f->cur.stride[0],
 1358|  37.5k|                                                         cbw4 - (furthest_r >> ss_hor),
 1359|  37.5k|                                                         cbh4 - (furthest_b >> ss_ver),
 1360|  37.5k|                                                         cbw4 * 4, cbh4 * 4);
 1361|   112k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1361:34): [True: 75.0k, False: 37.5k]
  ------------------
 1362|  75.0k|                    if (!b->cfl_alpha[pl]) continue;
  ------------------
  |  Branch (1362:25): [True: 16.7k, False: 58.3k]
  ------------------
 1363|  58.3k|                    int angle = 0;
 1364|  58.3k|                    const pixel *top_sb_edge = NULL;
 1365|  58.3k|                    if (!((t->by & ~ss_ver) & (f->sb_step - 1))) {
  ------------------
  |  Branch (1365:25): [True: 7.27k, False: 51.0k]
  ------------------
 1366|  7.27k|                        top_sb_edge = f->ipred_edge[pl + 1];
 1367|  7.27k|                        const int sby = t->by >> f->sb_shift;
 1368|  7.27k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1369|  7.27k|                    }
 1370|  58.3k|                    const int xpos = t->bx >> ss_hor, ypos = t->by >> ss_ver;
 1371|  58.3k|                    const int xstart = ts->tiling.col_start >> ss_hor;
 1372|  58.3k|                    const int ystart = ts->tiling.row_start >> ss_ver;
 1373|  58.3k|                    const enum IntraPredMode m =
 1374|  58.3k|                        bytefn(dav1d_prepare_intra_edges)(xpos, xpos > xstart,
  ------------------
  |  |   87|  58.3k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  58.3k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1375|  58.3k|                                                          ypos, ypos > ystart,
 1376|  58.3k|                                                          ts->tiling.col_end >> ss_hor,
 1377|  58.3k|                                                          ts->tiling.row_end >> ss_ver,
 1378|  58.3k|                                                          0, uv_dst[pl], stride,
 1379|  58.3k|                                                          top_sb_edge, DC_PRED, &angle,
 1380|  58.3k|                                                          uv_t_dim->w, uv_t_dim->h, 0,
 1381|  58.3k|                                                          edge HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  58.3k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1382|  58.3k|                    dsp->ipred.cfl_pred[m](uv_dst[pl], stride, edge,
 1383|  58.3k|                                           uv_t_dim->w * 4,
 1384|  58.3k|                                           uv_t_dim->h * 4,
 1385|  58.3k|                                           ac, b->cfl_alpha[pl]
 1386|  58.3k|                                           HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  58.3k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1387|  58.3k|                }
 1388|  37.5k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  37.5k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 37.5k]
  |  |  ------------------
  |  |   35|  37.5k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  37.5k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1389|      0|                    ac_dump(ac, 4*cbw4, 4*cbh4, "ac");
 1390|      0|                    hex_dump(uv_dst[0], stride, cbw4 * 4, cbh4 * 4, "u-cfl-pred");
 1391|      0|                    hex_dump(uv_dst[1], stride, cbw4 * 4, cbh4 * 4, "v-cfl-pred");
 1392|      0|                }
 1393|   628k|            } else if (b->pal_sz[1]) {
  ------------------
  |  Branch (1393:24): [True: 1.96k, False: 626k]
  ------------------
 1394|  1.96k|                const ptrdiff_t uv_dstoff = 4 * ((t->bx >> ss_hor) +
 1395|  1.96k|                                              (t->by >> ss_ver) * PXSTRIDE(f->cur.stride[1]));
 1396|  1.96k|                const pixel (*pal)[8];
 1397|  1.96k|                const uint8_t *pal_idx;
 1398|  1.96k|                if (t->frame_thread.pass) {
  ------------------
  |  Branch (1398:21): [True: 1.96k, False: 0]
  ------------------
 1399|  1.96k|                    const int p = t->frame_thread.pass & 1;
 1400|  1.96k|                    assert(ts->frame_thread[p].pal_idx);
  ------------------
  |  Branch (1400:21): [True: 1.96k, False: 0]
  ------------------
 1401|  1.96k|                    pal = f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 1402|  1.96k|                                              ((t->bx >> 1) + (t->by & 1))];
 1403|  1.96k|                    pal_idx = ts->frame_thread[p].pal_idx;
 1404|  1.96k|                    ts->frame_thread[p].pal_idx += cbw4 * cbh4 * 8;
 1405|  1.96k|                } else {
 1406|      0|                    pal = bytefn(t->scratch.pal);
  ------------------
  |  |   87|      0|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|      0|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1407|      0|                    pal_idx = t->scratch.pal_idx_uv;
 1408|      0|                }
 1409|       |
 1410|  1.96k|                f->dsp->ipred.pal_pred(((pixel *) f->cur.data[1]) + uv_dstoff,
 1411|  1.96k|                                       f->cur.stride[1], pal[1],
 1412|  1.96k|                                       pal_idx, cbw4 * 4, cbh4 * 4);
 1413|  1.96k|                f->dsp->ipred.pal_pred(((pixel *) f->cur.data[2]) + uv_dstoff,
 1414|  1.96k|                                       f->cur.stride[1], pal[2],
 1415|  1.96k|                                       pal_idx, cbw4 * 4, cbh4 * 4);
 1416|  1.96k|                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  1.96k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.96k]
  |  |  ------------------
  |  |   35|  1.96k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.96k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1417|      0|                    hex_dump(((pixel *) f->cur.data[1]) + uv_dstoff,
 1418|      0|                             PXSTRIDE(f->cur.stride[1]),
 1419|      0|                             cbw4 * 4, cbh4 * 4, "u-pal-pred");
 1420|      0|                    hex_dump(((pixel *) f->cur.data[2]) + uv_dstoff,
 1421|      0|                             PXSTRIDE(f->cur.stride[1]),
 1422|      0|                             cbw4 * 4, cbh4 * 4, "v-pal-pred");
 1423|      0|                }
 1424|  1.96k|            }
 1425|       |
 1426|   666k|            const int sm_uv_fl = sm_uv_flag(t->a, cbx4) |
 1427|   666k|                                 sm_uv_flag(&t->l, cby4);
 1428|   666k|            const int uv_sb_has_tr =
 1429|   666k|                ((init_x + 16) >> ss_hor) < cw4 ? 1 : init_y ? 0 :
  ------------------
  |  Branch (1429:17): [True: 127k, False: 538k]
  |  Branch (1429:55): [True: 68.2k, False: 469k]
  ------------------
 1430|   538k|                intra_edge_flags & (EDGE_I420_TOP_HAS_RIGHT >> (f->cur.p.layout - 1));
 1431|   666k|            const int uv_sb_has_bl =
 1432|   666k|                init_x ? 0 : ((init_y + 16) >> ss_ver) < ch4 ? 1 :
  ------------------
  |  Branch (1432:17): [True: 127k, False: 538k]
  |  Branch (1432:30): [True: 68.2k, False: 469k]
  ------------------
 1433|   538k|                intra_edge_flags & (EDGE_I420_LEFT_HAS_BOTTOM >> (f->cur.p.layout - 1));
 1434|   666k|            const int sub_cw4 = imin(cw4, (init_x + 16) >> ss_hor);
 1435|  1.99M|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1435:30): [True: 1.33M, False: 667k]
  ------------------
 1436|  3.53M|                for (y = init_y >> ss_ver, t->by += init_y; y < sub_ch4;
  ------------------
  |  Branch (1436:61): [True: 2.20M, False: 1.33M]
  ------------------
 1437|  2.20M|                     y += uv_t_dim->h, t->by += uv_t_dim->h << ss_ver)
 1438|  2.20M|                {
 1439|  2.20M|                    pixel *dst = ((pixel *) f->cur.data[1 + pl]) +
 1440|  2.20M|                                   4 * ((t->by >> ss_ver) * PXSTRIDE(stride) +
 1441|  2.20M|                                        ((t->bx + init_x) >> ss_hor));
 1442|  12.4M|                    for (x = init_x >> ss_hor, t->bx += init_x; x < sub_cw4;
  ------------------
  |  Branch (1442:65): [True: 10.2M, False: 2.20M]
  ------------------
 1443|  10.2M|                         x += uv_t_dim->w, t->bx += uv_t_dim->w << ss_hor)
 1444|  10.2M|                    {
 1445|  10.2M|                        if ((b->uv_mode == CFL_PRED && b->cfl_alpha[pl]) ||
  ------------------
  |  Branch (1445:30): [True: 75.0k, False: 10.1M]
  |  Branch (1445:56): [True: 58.3k, False: 16.7k]
  ------------------
 1446|  10.1M|                            b->pal_sz[1])
  ------------------
  |  Branch (1446:29): [True: 9.32k, False: 10.1M]
  ------------------
 1447|  68.3k|                        {
 1448|  68.3k|                            goto skip_uv_pred;
 1449|  68.3k|                        }
 1450|       |
 1451|  10.1M|                        int angle = b->uv_angle;
 1452|       |                        // this probably looks weird because we're using
 1453|       |                        // luma flags in a chroma loop, but that's because
 1454|       |                        // prepare_intra_edges() expects luma flags as input
 1455|  10.1M|                        const enum EdgeFlags edge_flags =
 1456|  10.1M|                            (((y > (init_y >> ss_ver) || !uv_sb_has_tr) &&
  ------------------
  |  Branch (1456:32): [True: 8.31M, False: 1.86M]
  |  Branch (1456:58): [True: 457k, False: 1.40M]
  ------------------
 1457|  8.77M|                              (x + uv_t_dim->w >= sub_cw4)) ?
  ------------------
  |  Branch (1457:31): [True: 1.19M, False: 7.58M]
  ------------------
 1458|  8.98M|                                 0 : EDGE_I444_TOP_HAS_RIGHT) |
 1459|  10.1M|                            ((x > (init_x >> ss_hor) ||
  ------------------
  |  Branch (1459:31): [True: 8.03M, False: 2.14M]
  ------------------
 1460|  2.14M|                              (!uv_sb_has_bl && y + uv_t_dim->h >= sub_ch4)) ?
  ------------------
  |  Branch (1460:32): [True: 1.11M, False: 1.02M]
  |  Branch (1460:49): [True: 662k, False: 456k]
  ------------------
 1461|  8.70M|                                 0 : EDGE_I444_LEFT_HAS_BOTTOM);
 1462|  10.1M|                        const pixel *top_sb_edge = NULL;
 1463|  10.1M|                        if (!((t->by & ~ss_ver) & (f->sb_step - 1))) {
  ------------------
  |  Branch (1463:29): [True: 934k, False: 9.24M]
  ------------------
 1464|   934k|                            top_sb_edge = f->ipred_edge[1 + pl];
 1465|   934k|                            const int sby = t->by >> f->sb_shift;
 1466|   934k|                            top_sb_edge += f->sb128w * 128 * (sby - 1);
 1467|   934k|                        }
 1468|  10.1M|                        const enum IntraPredMode uv_mode =
 1469|  10.1M|                             b->uv_mode == CFL_PRED ? DC_PRED : b->uv_mode;
  ------------------
  |  Branch (1469:30): [True: 16.7k, False: 10.1M]
  ------------------
 1470|  10.1M|                        const int xpos = t->bx >> ss_hor, ypos = t->by >> ss_ver;
 1471|  10.1M|                        const int xstart = ts->tiling.col_start >> ss_hor;
 1472|  10.1M|                        const int ystart = ts->tiling.row_start >> ss_ver;
 1473|  10.1M|                        const enum IntraPredMode m =
 1474|  10.1M|                            bytefn(dav1d_prepare_intra_edges)(xpos, xpos > xstart,
  ------------------
  |  |   87|  10.1M|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  10.1M|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1475|  10.1M|                                                              ypos, ypos > ystart,
 1476|  10.1M|                                                              ts->tiling.col_end >> ss_hor,
 1477|  10.1M|                                                              ts->tiling.row_end >> ss_ver,
 1478|  10.1M|                                                              edge_flags, dst, stride,
 1479|  10.1M|                                                              top_sb_edge, uv_mode,
 1480|  10.1M|                                                              &angle, uv_t_dim->w,
 1481|  10.1M|                                                              uv_t_dim->h,
 1482|  10.1M|                                                              f->seq_hdr->intra_edge_filter,
 1483|  10.1M|                                                              edge HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  10.1M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1484|  10.1M|                        angle |= intra_edge_filter_flag;
 1485|  10.1M|                        dsp->ipred.intra_pred[m](dst, stride, edge,
 1486|  10.1M|                                                 uv_t_dim->w * 4,
 1487|  10.1M|                                                 uv_t_dim->h * 4,
 1488|  10.1M|                                                 angle | sm_uv_fl,
 1489|  10.1M|                                                 (4 * f->bw + ss_hor -
 1490|  10.1M|                                                  4 * (t->bx & ~ss_hor)) >> ss_hor,
 1491|  10.1M|                                                 (4 * f->bh + ss_ver -
 1492|  10.1M|                                                  4 * (t->by & ~ss_ver)) >> ss_ver
 1493|  10.1M|                                                 HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  10.1M|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1494|  10.1M|                        if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  10.1M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 10.1M]
  |  |  ------------------
  |  |   35|  10.1M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  10.1M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                      if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1495|      0|                            hex_dump(edge - uv_t_dim->h * 4, uv_t_dim->h * 4,
 1496|      0|                                     uv_t_dim->h * 4, 2, "l");
 1497|      0|                            hex_dump(edge, 0, 1, 1, "tl");
 1498|      0|                            hex_dump(edge + 1, uv_t_dim->w * 4,
 1499|      0|                                     uv_t_dim->w * 4, 2, "t");
 1500|      0|                            hex_dump(dst, stride, uv_t_dim->w * 4,
 1501|      0|                                     uv_t_dim->h * 4, pl ? "v-intra-pred" : "u-intra-pred");
  ------------------
  |  Branch (1501:55): [True: 0, False: 0]
  ------------------
 1502|      0|                        }
 1503|       |
 1504|  10.2M|                    skip_uv_pred: {}
 1505|  10.2M|                        if (!b->skip) {
  ------------------
  |  Branch (1505:29): [True: 1.02M, False: 9.21M]
  ------------------
 1506|  1.02M|                            enum TxfmType txtp;
 1507|  1.02M|                            int eob;
 1508|  1.02M|                            coef *cf;
 1509|  1.02M|                            if (t->frame_thread.pass) {
  ------------------
  |  Branch (1509:33): [True: 1.02M, False: 18.4E]
  ------------------
 1510|  1.02M|                                const int p = t->frame_thread.pass & 1;
 1511|  1.02M|                                const int cbi = *ts->frame_thread[p].cbi++;
 1512|  1.02M|                                cf = ts->frame_thread[p].cf;
 1513|  1.02M|                                ts->frame_thread[p].cf += uv_t_dim->w * uv_t_dim->h * 16;
 1514|  1.02M|                                eob  = cbi >> 5;
 1515|  1.02M|                                txtp = cbi & 0x1f;
 1516|  18.4E|                            } else {
 1517|  18.4E|                                uint8_t cf_ctx;
 1518|  18.4E|                                cf = bitfn(t->cf);
  ------------------
  |  |   77|  18.4E|#define bitfn(x) x##_16bpc
  ------------------
 1519|  18.4E|                                eob = decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
 1520|  18.4E|                                                   &t->l.ccoef[pl][cby4 + y],
 1521|  18.4E|                                                   b->uvtx, bs, b, 1, 1 + pl, cf,
 1522|  18.4E|                                                   &txtp, &cf_ctx);
 1523|  18.4E|                                if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  18.4E|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 18.4E]
  |  |  ------------------
  |  |   35|  18.4E|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  18.4E|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1524|      0|                                    printf("Post-uv-cf-blk[pl=%d,tx=%d,"
 1525|      0|                                           "txtp=%d,eob=%d]: r=%d [x=%d,cbx4=%d]\n",
 1526|      0|                                           pl, b->uvtx, txtp, eob, ts->msac.rng, x, cbx4);
 1527|  18.4E|                                int ctw = imin(uv_t_dim->w, (f->bw - t->bx + ss_hor) >> ss_hor);
 1528|  18.4E|                                int cth = imin(uv_t_dim->h, (f->bh - t->by + ss_ver) >> ss_ver);
 1529|  18.4E|                                dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
 1530|  18.4E|                                dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
 1531|  18.4E|                            }
 1532|  1.02M|                            if (eob >= 0) {
  ------------------
  |  Branch (1532:33): [True: 230k, False: 798k]
  ------------------
 1533|   230k|                                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   230k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 230k]
  |  |  ------------------
  |  |   35|   230k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   230k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1534|      0|                                    coef_dump(cf, uv_t_dim->h * 4,
 1535|      0|                                              uv_t_dim->w * 4, 3, "dq");
 1536|   230k|                                dsp->itx.itxfm_add[b->uvtx]
 1537|   230k|                                                  [txtp](dst, stride,
 1538|   230k|                                                         cf, eob HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|   230k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1539|   230k|                                if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|   230k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 230k]
  |  |  ------------------
  |  |   35|   230k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|   230k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                              if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1540|      0|                                    hex_dump(dst, stride, uv_t_dim->w * 4,
 1541|      0|                                             uv_t_dim->h * 4, "recon");
 1542|   230k|                            }
 1543|  9.21M|                        } else if (!t->frame_thread.pass) {
  ------------------
  |  Branch (1543:36): [True: 0, False: 9.21M]
  ------------------
 1544|      0|                            dav1d_memset_pow2[uv_t_dim->lw](&t->a->ccoef[pl][cbx4 + x], 0x40);
 1545|      0|                            dav1d_memset_pow2[uv_t_dim->lh](&t->l.ccoef[pl][cby4 + y], 0x40);
 1546|      0|                        }
 1547|  10.2M|                        dst += uv_t_dim->w * 4;
 1548|  10.2M|                    }
 1549|  2.20M|                    t->bx -= x << ss_hor;
 1550|  2.20M|                }
 1551|  1.33M|                t->by -= y << ss_ver;
 1552|  1.33M|            }
 1553|   666k|        }
 1554|  1.03M|    }
 1555|   887k|}
dav1d_recon_b_inter_16bpc:
 1559|  1.04M|{
 1560|  1.04M|    Dav1dTileState *const ts = t->ts;
 1561|  1.04M|    const Dav1dFrameContext *const f = t->f;
 1562|  1.04M|    const Dav1dDSPContext *const dsp = f->dsp;
 1563|  1.04M|    const int bx4 = t->bx & 31, by4 = t->by & 31;
 1564|  1.04M|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1565|  1.04M|    const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 1566|  1.04M|    const int cbx4 = bx4 >> ss_hor, cby4 = by4 >> ss_ver;
 1567|  1.04M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
 1568|  1.04M|    const int bw4 = b_dim[0], bh4 = b_dim[1];
 1569|  1.04M|    const int w4 = imin(bw4, f->bw - t->bx), h4 = imin(bh4, f->bh - t->by);
 1570|  1.04M|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400 &&
  ------------------
  |  Branch (1570:28): [True: 285k, False: 754k]
  ------------------
 1571|   285k|                           (bw4 > ss_hor || t->bx & 1) &&
  ------------------
  |  Branch (1571:29): [True: 256k, False: 28.4k]
  |  Branch (1571:45): [True: 13.7k, False: 14.7k]
  ------------------
 1572|   270k|                           (bh4 > ss_ver || t->by & 1);
  ------------------
  |  Branch (1572:29): [True: 240k, False: 29.6k]
  |  Branch (1572:45): [True: 14.2k, False: 15.3k]
  ------------------
 1573|  1.04M|    const int chr_layout_idx = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I400 ? 0 :
  ------------------
  |  Branch (1573:32): [True: 754k, False: 285k]
  ------------------
 1574|  1.04M|                               DAV1D_PIXEL_LAYOUT_I444 - f->cur.p.layout;
 1575|  1.04M|    int res;
 1576|       |
 1577|       |    // prediction
 1578|  1.04M|    const int cbh4 = (bh4 + ss_ver) >> ss_ver, cbw4 = (bw4 + ss_hor) >> ss_hor;
 1579|  1.04M|    pixel *dst = ((pixel *) f->cur.data[0]) +
 1580|  1.04M|        4 * (t->by * PXSTRIDE(f->cur.stride[0]) + t->bx);
 1581|  1.04M|    const ptrdiff_t uvdstoff =
 1582|  1.04M|        4 * ((t->bx >> ss_hor) + (t->by >> ss_ver) * PXSTRIDE(f->cur.stride[1]));
 1583|  1.04M|    if (IS_KEY_OR_INTRA(f->frame_hdr)) {
  ------------------
  |  |   43|  1.04M|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  1.04M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  |  |  |  Branch (43:5): [True: 6.97k, False: 1.03M]
  |  |  ------------------
  ------------------
 1584|       |        // intrabc
 1585|  6.97k|        assert(!f->frame_hdr->super_res.enabled);
  ------------------
  |  Branch (1585:9): [True: 6.97k, False: 0]
  ------------------
 1586|  6.97k|        res = mc(t, dst, NULL, f->cur.stride[0], bw4, bh4, t->bx, t->by, 0,
 1587|  6.97k|                 b->mv[0], &f->sr_cur, 0 /* unused */, FILTER_2D_BILINEAR);
 1588|  6.97k|        if (res) return res;
  ------------------
  |  Branch (1588:13): [True: 0, False: 6.97k]
  ------------------
 1589|  16.0k|        if (has_chroma) for (int pl = 1; pl < 3; pl++) {
  ------------------
  |  Branch (1589:13): [True: 5.35k, False: 1.62k]
  |  Branch (1589:42): [True: 10.7k, False: 5.35k]
  ------------------
 1590|  10.7k|            res = mc(t, ((pixel *)f->cur.data[pl]) + uvdstoff, NULL, f->cur.stride[1],
 1591|  10.7k|                     bw4 << (bw4 == ss_hor), bh4 << (bh4 == ss_ver),
 1592|  10.7k|                     t->bx & ~ss_hor, t->by & ~ss_ver, pl, b->mv[0],
 1593|  10.7k|                     &f->sr_cur, 0 /* unused */, FILTER_2D_BILINEAR);
 1594|  10.7k|            if (res) return res;
  ------------------
  |  Branch (1594:17): [True: 0, False: 10.7k]
  ------------------
 1595|  10.7k|        }
 1596|  1.03M|    } else if (b->comp_type == COMP_INTER_NONE) {
  ------------------
  |  Branch (1596:16): [True: 981k, False: 51.8k]
  ------------------
 1597|   981k|        const Dav1dThreadPicture *const refp = &f->refp[b->ref[0]];
 1598|   981k|        const enum Filter2d filter_2d = b->filter2d;
 1599|       |
 1600|   981k|        if (imin(bw4, bh4) > 1 &&
  ------------------
  |  Branch (1600:13): [True: 884k, False: 96.9k]
  ------------------
 1601|   884k|            ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1601:15): [True: 758k, False: 125k]
  |  Branch (1601:44): [True: 123k, False: 634k]
  ------------------
 1602|   760k|             (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1602:15): [True: 37.2k, False: 723k]
  |  Branch (1602:44): [True: 32.9k, False: 4.29k]
  ------------------
 1603|   156k|        {
 1604|   156k|            res = warp_affine(t, dst, NULL, f->cur.stride[0], b_dim, 0, refp,
 1605|   156k|                              b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1605:31): [True: 32.9k, False: 123k]
  ------------------
 1606|   156k|                                  &f->frame_hdr->gmv[b->ref[0]]);
 1607|   156k|            if (res) return res;
  ------------------
  |  Branch (1607:17): [True: 0, False: 156k]
  ------------------
 1608|   824k|        } else {
 1609|   824k|            res = mc(t, dst, NULL, f->cur.stride[0],
 1610|   824k|                     bw4, bh4, t->bx, t->by, 0, b->mv[0], refp, b->ref[0], filter_2d);
 1611|   824k|            if (res) return res;
  ------------------
  |  Branch (1611:17): [True: 0, False: 824k]
  ------------------
 1612|   824k|            if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1612:17): [True: 50.0k, False: 774k]
  ------------------
 1613|  50.0k|                res = obmc(t, dst, f->cur.stride[0], b_dim, 0, bx4, by4, w4, h4);
 1614|  50.0k|                if (res) return res;
  ------------------
  |  Branch (1614:21): [True: 0, False: 50.0k]
  ------------------
 1615|  50.0k|            }
 1616|   824k|        }
 1617|   981k|        if (b->interintra_type) {
  ------------------
  |  Branch (1617:13): [True: 11.0k, False: 970k]
  ------------------
 1618|  11.0k|            pixel *const tl_edge = bitfn(t->scratch.edge) + 32;
  ------------------
  |  |   77|  11.0k|#define bitfn(x) x##_16bpc
  ------------------
 1619|  11.0k|            enum IntraPredMode m = b->interintra_mode == II_SMOOTH_PRED ?
  ------------------
  |  Branch (1619:36): [True: 2.00k, False: 9.01k]
  ------------------
 1620|  9.01k|                                   SMOOTH_PRED : b->interintra_mode;
 1621|  11.0k|            pixel *const tmp = bitfn(t->scratch.interintra);
  ------------------
  |  |   77|  11.0k|#define bitfn(x) x##_16bpc
  ------------------
 1622|  11.0k|            int angle = 0;
 1623|  11.0k|            const pixel *top_sb_edge = NULL;
 1624|  11.0k|            if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1624:17): [True: 1.53k, False: 9.48k]
  ------------------
 1625|  1.53k|                top_sb_edge = f->ipred_edge[0];
 1626|  1.53k|                const int sby = t->by >> f->sb_shift;
 1627|  1.53k|                top_sb_edge += f->sb128w * 128 * (sby - 1);
 1628|  1.53k|            }
 1629|  11.0k|            m = bytefn(dav1d_prepare_intra_edges)(t->bx, t->bx > ts->tiling.col_start,
  ------------------
  |  |   87|  11.0k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  11.0k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1630|  11.0k|                                                  t->by, t->by > ts->tiling.row_start,
 1631|  11.0k|                                                  ts->tiling.col_end, ts->tiling.row_end,
 1632|  11.0k|                                                  0, dst, f->cur.stride[0], top_sb_edge,
 1633|  11.0k|                                                  m, &angle, bw4, bh4, 0, tl_edge
 1634|  11.0k|                                                  HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  11.0k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1635|  11.0k|            dsp->ipred.intra_pred[m](tmp, 4 * bw4 * sizeof(pixel),
 1636|  11.0k|                                     tl_edge, bw4 * 4, bh4 * 4, 0, 0, 0
 1637|  11.0k|                                     HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  11.0k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1638|  11.0k|            dsp->mc.blend(dst, f->cur.stride[0], tmp,
 1639|  11.0k|                          bw4 * 4, bh4 * 4, II_MASK(0, bs, b));
  ------------------
  |  |   83|  11.0k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   84|  11.0k|    (size_t)((b)->interintra_type == INTER_INTRA_BLEND ? \
  |  |  ------------------
  |  |  |  Branch (84:14): [True: 8.38k, False: 2.63k]
  |  |  ------------------
  |  |   85|  11.0k|    dav1d_masks.offsets[c][(bs)-BS_32x32].ii[(b)->interintra_mode] : \
  |  |   86|  11.0k|    dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[0][(b)->wedge_idx]) * 8))
  ------------------
 1640|  11.0k|        }
 1641|       |
 1642|   981k|        if (!has_chroma) goto skip_inter_chroma_pred;
  ------------------
  |  Branch (1642:13): [True: 751k, False: 229k]
  ------------------
 1643|       |
 1644|       |        // sub8x8 derivation
 1645|   229k|        int is_sub8x8 = bw4 == ss_hor || bh4 == ss_ver;
  ------------------
  |  Branch (1645:25): [True: 9.70k, False: 220k]
  |  Branch (1645:42): [True: 11.5k, False: 208k]
  ------------------
 1646|   229k|        refmvs_block *const *r;
 1647|   229k|        if (is_sub8x8) {
  ------------------
  |  Branch (1647:13): [True: 21.5k, False: 208k]
  ------------------
 1648|  21.5k|            assert(ss_hor == 1);
  ------------------
  |  Branch (1648:13): [True: 21.5k, False: 0]
  ------------------
 1649|  21.5k|            r = &t->rt.r[(t->by & 31) + 5];
 1650|  21.5k|            if (bw4 == 1) is_sub8x8 &= r[0][t->bx - 1].ref.ref[0] > 0;
  ------------------
  |  Branch (1650:17): [True: 9.96k, False: 11.5k]
  ------------------
 1651|  21.5k|            if (bh4 == ss_ver) is_sub8x8 &= r[-1][t->bx].ref.ref[0] > 0;
  ------------------
  |  Branch (1651:17): [True: 13.7k, False: 7.78k]
  ------------------
 1652|  21.5k|            if (bw4 == 1 && bh4 == ss_ver)
  ------------------
  |  Branch (1652:17): [True: 9.96k, False: 11.5k]
  |  Branch (1652:29): [True: 2.18k, False: 7.78k]
  ------------------
 1653|  2.18k|                is_sub8x8 &= r[-1][t->bx - 1].ref.ref[0] > 0;
 1654|  21.5k|        }
 1655|       |
 1656|       |        // chroma prediction
 1657|   229k|        if (is_sub8x8) {
  ------------------
  |  Branch (1657:13): [True: 16.8k, False: 213k]
  ------------------
 1658|  16.8k|            assert(ss_hor == 1);
  ------------------
  |  Branch (1658:13): [True: 16.8k, False: 0]
  ------------------
 1659|  16.8k|            ptrdiff_t h_off = 0, v_off = 0;
 1660|  16.8k|            if (bw4 == 1 && bh4 == ss_ver) {
  ------------------
  |  Branch (1660:17): [True: 7.81k, False: 9.07k]
  |  Branch (1660:29): [True: 1.54k, False: 6.27k]
  ------------------
 1661|  4.62k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1661:34): [True: 3.08k, False: 1.54k]
  ------------------
 1662|  3.08k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1663|  3.08k|                             NULL, f->cur.stride[1],
 1664|  3.08k|                             bw4, bh4, t->bx - 1, t->by - 1, 1 + pl,
 1665|  3.08k|                             r[-1][t->bx - 1].mv.mv[0],
 1666|  3.08k|                             &f->refp[r[-1][t->bx - 1].ref.ref[0] - 1],
 1667|  3.08k|                             r[-1][t->bx - 1].ref.ref[0] - 1,
 1668|  3.08k|                             t->frame_thread.pass != 2 ? t->tl_4x4_filter :
  ------------------
  |  Branch (1668:30): [True: 0, False: 3.08k]
  ------------------
 1669|  3.08k|                                 f->frame_thread.b[((t->by - 1) * f->b4_stride) + t->bx - 1].filter2d);
 1670|  3.08k|                    if (res) return res;
  ------------------
  |  Branch (1670:25): [True: 0, False: 3.08k]
  ------------------
 1671|  3.08k|                }
 1672|  1.54k|                v_off = 2 * PXSTRIDE(f->cur.stride[1]);
 1673|  1.54k|                h_off = 2;
 1674|  1.54k|            }
 1675|  16.8k|            if (bw4 == 1) {
  ------------------
  |  Branch (1675:17): [True: 7.81k, False: 9.07k]
  ------------------
 1676|  7.81k|                const enum Filter2d left_filter_2d =
 1677|  7.81k|                    dav1d_filter_2d[t->l.filter[1][by4]][t->l.filter[0][by4]];
 1678|  23.4k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1678:34): [True: 15.6k, False: 7.81k]
  ------------------
 1679|  15.6k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + v_off, NULL,
 1680|  15.6k|                             f->cur.stride[1], bw4, bh4, t->bx - 1,
 1681|  15.6k|                             t->by, 1 + pl, r[0][t->bx - 1].mv.mv[0],
 1682|  15.6k|                             &f->refp[r[0][t->bx - 1].ref.ref[0] - 1],
 1683|  15.6k|                             r[0][t->bx - 1].ref.ref[0] - 1,
 1684|  15.6k|                             t->frame_thread.pass != 2 ? left_filter_2d :
  ------------------
  |  Branch (1684:30): [True: 0, False: 15.6k]
  ------------------
 1685|  15.6k|                                 f->frame_thread.b[(t->by * f->b4_stride) + t->bx - 1].filter2d);
 1686|  15.6k|                    if (res) return res;
  ------------------
  |  Branch (1686:25): [True: 0, False: 15.6k]
  ------------------
 1687|  15.6k|                }
 1688|  7.81k|                h_off = 2;
 1689|  7.81k|            }
 1690|  16.8k|            if (bh4 == ss_ver) {
  ------------------
  |  Branch (1690:17): [True: 10.6k, False: 6.27k]
  ------------------
 1691|  10.6k|                const enum Filter2d top_filter_2d =
 1692|  10.6k|                    dav1d_filter_2d[t->a->filter[1][bx4]][t->a->filter[0][bx4]];
 1693|  31.8k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1693:34): [True: 21.2k, False: 10.6k]
  ------------------
 1694|  21.2k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + h_off, NULL,
 1695|  21.2k|                             f->cur.stride[1], bw4, bh4, t->bx, t->by - 1,
 1696|  21.2k|                             1 + pl, r[-1][t->bx].mv.mv[0],
 1697|  21.2k|                             &f->refp[r[-1][t->bx].ref.ref[0] - 1],
 1698|  21.2k|                             r[-1][t->bx].ref.ref[0] - 1,
 1699|  21.2k|                             t->frame_thread.pass != 2 ? top_filter_2d :
  ------------------
  |  Branch (1699:30): [True: 0, False: 21.2k]
  ------------------
 1700|  21.2k|                                 f->frame_thread.b[((t->by - 1) * f->b4_stride) + t->bx].filter2d);
 1701|  21.2k|                    if (res) return res;
  ------------------
  |  Branch (1701:25): [True: 0, False: 21.2k]
  ------------------
 1702|  21.2k|                }
 1703|  10.6k|                v_off = 2 * PXSTRIDE(f->cur.stride[1]);
 1704|  10.6k|            }
 1705|  50.6k|            for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1705:30): [True: 33.7k, False: 16.8k]
  ------------------
 1706|  33.7k|                res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff + h_off + v_off, NULL, f->cur.stride[1],
 1707|  33.7k|                         bw4, bh4, t->bx, t->by, 1 + pl, b->mv[0],
 1708|  33.7k|                         refp, b->ref[0], filter_2d);
 1709|  33.7k|                if (res) return res;
  ------------------
  |  Branch (1709:21): [True: 0, False: 33.7k]
  ------------------
 1710|  33.7k|            }
 1711|   213k|        } else {
 1712|   213k|            if (imin(cbw4, cbh4) > 1 &&
  ------------------
  |  Branch (1712:17): [True: 128k, False: 84.2k]
  ------------------
 1713|   128k|                ((b->inter_mode == GLOBALMV && f->gmv_warp_allowed[b->ref[0]]) ||
  ------------------
  |  Branch (1713:19): [True: 84.6k, False: 44.1k]
  |  Branch (1713:48): [True: 11.2k, False: 73.4k]
  ------------------
 1714|   117k|                 (b->motion_mode == MM_WARP && t->warpmv.type > DAV1D_WM_TYPE_TRANSLATION)))
  ------------------
  |  Branch (1714:19): [True: 14.2k, False: 103k]
  |  Branch (1714:48): [True: 13.7k, False: 522]
  ------------------
 1715|  24.9k|            {
 1716|  74.8k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1716:34): [True: 49.9k, False: 24.9k]
  ------------------
 1717|  49.9k|                    res = warp_affine(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff, NULL,
 1718|  49.9k|                                      f->cur.stride[1], b_dim, 1 + pl, refp,
 1719|  49.9k|                                      b->motion_mode == MM_WARP ? &t->warpmv :
  ------------------
  |  Branch (1719:39): [True: 27.5k, False: 22.4k]
  ------------------
 1720|  49.9k|                                          &f->frame_hdr->gmv[b->ref[0]]);
 1721|  49.9k|                    if (res) return res;
  ------------------
  |  Branch (1721:25): [True: 0, False: 49.9k]
  ------------------
 1722|  49.9k|                }
 1723|   188k|            } else {
 1724|   564k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1724:34): [True: 376k, False: 188k]
  ------------------
 1725|   376k|                    res = mc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1726|   376k|                             NULL, f->cur.stride[1],
 1727|   376k|                             bw4 << (bw4 == ss_hor), bh4 << (bh4 == ss_ver),
 1728|   376k|                             t->bx & ~ss_hor, t->by & ~ss_ver,
 1729|   376k|                             1 + pl, b->mv[0], refp, b->ref[0], filter_2d);
 1730|   376k|                    if (res) return res;
  ------------------
  |  Branch (1730:25): [True: 0, False: 376k]
  ------------------
 1731|   376k|                    if (b->motion_mode == MM_OBMC) {
  ------------------
  |  Branch (1731:25): [True: 68.3k, False: 308k]
  ------------------
 1732|  68.3k|                        res = obmc(t, ((pixel *) f->cur.data[1 + pl]) + uvdstoff,
 1733|  68.3k|                                   f->cur.stride[1], b_dim, 1 + pl, bx4, by4, w4, h4);
 1734|  68.3k|                        if (res) return res;
  ------------------
  |  Branch (1734:29): [True: 0, False: 68.3k]
  ------------------
 1735|  68.3k|                    }
 1736|   376k|                }
 1737|   188k|            }
 1738|   213k|            if (b->interintra_type) {
  ------------------
  |  Branch (1738:17): [True: 9.78k, False: 203k]
  ------------------
 1739|       |                // FIXME for 8x32 with 4:2:2 subsampling, this probably does
 1740|       |                // the wrong thing since it will select 4x16, not 4x32, as a
 1741|       |                // transform size...
 1742|  9.78k|                const uint8_t *const ii_mask = II_MASK(chr_layout_idx, bs, b);
  ------------------
  |  |   83|  9.78k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   84|  9.78k|    (size_t)((b)->interintra_type == INTER_INTRA_BLEND ? \
  |  |  ------------------
  |  |  |  Branch (84:14): [True: 7.47k, False: 2.31k]
  |  |  ------------------
  |  |   85|  9.78k|    dav1d_masks.offsets[c][(bs)-BS_32x32].ii[(b)->interintra_mode] : \
  |  |   86|  9.78k|    dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[0][(b)->wedge_idx]) * 8))
  ------------------
 1743|       |
 1744|  29.3k|                for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1744:34): [True: 19.5k, False: 9.78k]
  ------------------
 1745|  19.5k|                    pixel *const tmp = bitfn(t->scratch.interintra);
  ------------------
  |  |   77|  19.5k|#define bitfn(x) x##_16bpc
  ------------------
 1746|  19.5k|                    pixel *const tl_edge = bitfn(t->scratch.edge) + 32;
  ------------------
  |  |   77|  19.5k|#define bitfn(x) x##_16bpc
  ------------------
 1747|  19.5k|                    enum IntraPredMode m =
 1748|  19.5k|                        b->interintra_mode == II_SMOOTH_PRED ?
  ------------------
  |  Branch (1748:25): [True: 3.42k, False: 16.1k]
  ------------------
 1749|  16.1k|                        SMOOTH_PRED : b->interintra_mode;
 1750|  19.5k|                    int angle = 0;
 1751|  19.5k|                    pixel *const uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff;
 1752|  19.5k|                    const pixel *top_sb_edge = NULL;
 1753|  19.5k|                    if (!(t->by & (f->sb_step - 1))) {
  ------------------
  |  Branch (1753:25): [True: 2.53k, False: 17.0k]
  ------------------
 1754|  2.53k|                        top_sb_edge = f->ipred_edge[pl + 1];
 1755|  2.53k|                        const int sby = t->by >> f->sb_shift;
 1756|  2.53k|                        top_sb_edge += f->sb128w * 128 * (sby - 1);
 1757|  2.53k|                    }
 1758|  19.5k|                    m = bytefn(dav1d_prepare_intra_edges)(t->bx >> ss_hor,
  ------------------
  |  |   87|  19.5k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  19.5k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 1759|  19.5k|                                                          (t->bx >> ss_hor) >
 1760|  19.5k|                                                              (ts->tiling.col_start >> ss_hor),
 1761|  19.5k|                                                          t->by >> ss_ver,
 1762|  19.5k|                                                          (t->by >> ss_ver) >
 1763|  19.5k|                                                              (ts->tiling.row_start >> ss_ver),
 1764|  19.5k|                                                          ts->tiling.col_end >> ss_hor,
 1765|  19.5k|                                                          ts->tiling.row_end >> ss_ver,
 1766|  19.5k|                                                          0, uvdst, f->cur.stride[1],
 1767|  19.5k|                                                          top_sb_edge, m,
 1768|  19.5k|                                                          &angle, cbw4, cbh4, 0, tl_edge
 1769|  19.5k|                                                          HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  19.5k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1770|  19.5k|                    dsp->ipred.intra_pred[m](tmp, cbw4 * 4 * sizeof(pixel),
 1771|  19.5k|                                             tl_edge, cbw4 * 4, cbh4 * 4, 0, 0, 0
 1772|  19.5k|                                             HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  19.5k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1773|  19.5k|                    dsp->mc.blend(uvdst, f->cur.stride[1], tmp,
 1774|  19.5k|                                  cbw4 * 4, cbh4 * 4, ii_mask);
 1775|  19.5k|                }
 1776|  9.78k|            }
 1777|   213k|        }
 1778|       |
 1779|   981k|    skip_inter_chroma_pred: {}
 1780|   981k|        t->tl_4x4_filter = filter_2d;
 1781|   981k|    } else {
 1782|  51.8k|        const enum Filter2d filter_2d = b->filter2d;
 1783|       |        // Maximum super block size is 128x128
 1784|  51.8k|        int16_t (*tmp)[128 * 128] = t->scratch.compinter;
 1785|  51.8k|        int jnt_weight;
 1786|  51.8k|        uint8_t *const seg_mask = t->scratch.seg_mask;
 1787|  51.8k|        const uint8_t *mask;
 1788|       |
 1789|   154k|        for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1789:25): [True: 103k, False: 51.8k]
  ------------------
 1790|   103k|            const Dav1dThreadPicture *const refp = &f->refp[b->ref[i]];
 1791|       |
 1792|   103k|            if (b->inter_mode == GLOBALMV_GLOBALMV && f->gmv_warp_allowed[b->ref[i]]) {
  ------------------
  |  Branch (1792:17): [True: 18.3k, False: 84.6k]
  |  Branch (1792:55): [True: 7.60k, False: 10.7k]
  ------------------
 1793|  7.60k|                res = warp_affine(t, NULL, tmp[i], bw4 * 4, b_dim, 0, refp,
 1794|  7.60k|                                  &f->frame_hdr->gmv[b->ref[i]]);
 1795|  7.60k|                if (res) return res;
  ------------------
  |  Branch (1795:21): [True: 0, False: 7.60k]
  ------------------
 1796|  95.4k|            } else {
 1797|  95.4k|                res = mc(t, NULL, tmp[i], 0, bw4, bh4, t->bx, t->by, 0,
 1798|  95.4k|                         b->mv[i], refp, b->ref[i], filter_2d);
 1799|  95.4k|                if (res) return res;
  ------------------
  |  Branch (1799:21): [True: 0, False: 95.4k]
  ------------------
 1800|  95.4k|            }
 1801|   103k|        }
 1802|  51.8k|        switch (b->comp_type) {
  ------------------
  |  Branch (1802:17): [True: 51.5k, False: 320]
  ------------------
 1803|  30.0k|        case COMP_INTER_AVG:
  ------------------
  |  Branch (1803:9): [True: 30.0k, False: 21.7k]
  ------------------
 1804|  30.0k|            dsp->mc.avg(dst, f->cur.stride[0], tmp[0], tmp[1],
 1805|  30.0k|                        bw4 * 4, bh4 * 4 HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  30.0k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1806|  30.0k|            break;
 1807|  8.59k|        case COMP_INTER_WEIGHTED_AVG:
  ------------------
  |  Branch (1807:9): [True: 8.59k, False: 43.2k]
  ------------------
 1808|  8.59k|            jnt_weight = f->jnt_weights[b->ref[0]][b->ref[1]];
 1809|  8.59k|            dsp->mc.w_avg(dst, f->cur.stride[0], tmp[0], tmp[1],
 1810|  8.59k|                          bw4 * 4, bh4 * 4, jnt_weight HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  8.59k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1811|  8.59k|            break;
 1812|  9.30k|        case COMP_INTER_SEG:
  ------------------
  |  Branch (1812:9): [True: 9.30k, False: 42.5k]
  ------------------
 1813|  9.30k|            dsp->mc.w_mask[chr_layout_idx](dst, f->cur.stride[0],
 1814|  9.30k|                                           tmp[b->mask_sign], tmp[!b->mask_sign],
 1815|  9.30k|                                           bw4 * 4, bh4 * 4, seg_mask,
 1816|  9.30k|                                           b->mask_sign HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  9.30k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1817|  9.30k|            mask = seg_mask;
 1818|  9.30k|            break;
 1819|  3.56k|        case COMP_INTER_WEDGE:
  ------------------
  |  Branch (1819:9): [True: 3.56k, False: 48.2k]
  ------------------
 1820|  3.56k|            mask = WEDGE_MASK(0, bs, 0, b->wedge_idx);
  ------------------
  |  |   89|  3.56k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   90|  3.56k|    (size_t)dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[sign][idx] * 8))
  ------------------
 1821|  3.56k|            dsp->mc.mask(dst, f->cur.stride[0],
 1822|  3.56k|                         tmp[b->mask_sign], tmp[!b->mask_sign],
 1823|  3.56k|                         bw4 * 4, bh4 * 4, mask HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  3.56k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1824|  3.56k|            if (has_chroma)
  ------------------
  |  Branch (1824:17): [True: 1.94k, False: 1.61k]
  ------------------
 1825|  1.94k|                mask = WEDGE_MASK(chr_layout_idx, bs, b->mask_sign, b->wedge_idx);
  ------------------
  |  |   89|  1.94k|    ((const uint8_t*)((uintptr_t)&dav1d_masks + \
  |  |   90|  1.94k|    (size_t)dav1d_masks.offsets[c][(bs)-BS_32x32].wedge[sign][idx] * 8))
  ------------------
 1826|  3.56k|            break;
 1827|  51.8k|        }
 1828|       |
 1829|       |        // chroma
 1830|  58.2k|        if (has_chroma) for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1830:13): [True: 19.4k, False: 32.0k]
  |  Branch (1830:42): [True: 38.8k, False: 19.4k]
  ------------------
 1831|   116k|            for (int i = 0; i < 2; i++) {
  ------------------
  |  Branch (1831:29): [True: 77.6k, False: 38.8k]
  ------------------
 1832|  77.6k|                const Dav1dThreadPicture *const refp = &f->refp[b->ref[i]];
 1833|  77.6k|                if (b->inter_mode == GLOBALMV_GLOBALMV &&
  ------------------
  |  Branch (1833:21): [True: 13.8k, False: 63.8k]
  ------------------
 1834|  13.8k|                    imin(cbw4, cbh4) > 1 && f->gmv_warp_allowed[b->ref[i]])
  ------------------
  |  Branch (1834:21): [True: 7.12k, False: 6.74k]
  |  Branch (1834:45): [True: 3.70k, False: 3.41k]
  ------------------
 1835|  3.70k|                {
 1836|  3.70k|                    res = warp_affine(t, NULL, tmp[i], bw4 * 4 >> ss_hor,
 1837|  3.70k|                                      b_dim, 1 + pl,
 1838|  3.70k|                                      refp, &f->frame_hdr->gmv[b->ref[i]]);
 1839|  3.70k|                    if (res) return res;
  ------------------
  |  Branch (1839:25): [True: 0, False: 3.70k]
  ------------------
 1840|  73.9k|                } else {
 1841|  73.9k|                    res = mc(t, NULL, tmp[i], 0, bw4, bh4, t->bx, t->by,
 1842|  73.9k|                             1 + pl, b->mv[i], refp, b->ref[i], filter_2d);
 1843|  73.9k|                    if (res) return res;
  ------------------
  |  Branch (1843:25): [True: 0, False: 73.9k]
  ------------------
 1844|  73.9k|                }
 1845|  77.6k|            }
 1846|  38.8k|            pixel *const uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff;
 1847|  38.8k|            switch (b->comp_type) {
  ------------------
  |  Branch (1847:21): [True: 38.8k, False: 18.4E]
  ------------------
 1848|  21.0k|            case COMP_INTER_AVG:
  ------------------
  |  Branch (1848:13): [True: 21.0k, False: 17.8k]
  ------------------
 1849|  21.0k|                dsp->mc.avg(uvdst, f->cur.stride[1], tmp[0], tmp[1],
 1850|  21.0k|                            bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver
 1851|  21.0k|                            HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  21.0k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1852|  21.0k|                break;
 1853|  3.11k|            case COMP_INTER_WEIGHTED_AVG:
  ------------------
  |  Branch (1853:13): [True: 3.11k, False: 35.7k]
  ------------------
 1854|  3.11k|                dsp->mc.w_avg(uvdst, f->cur.stride[1], tmp[0], tmp[1],
 1855|  3.11k|                              bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver, jnt_weight
 1856|  3.11k|                              HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  3.11k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1857|  3.11k|                break;
 1858|  3.89k|            case COMP_INTER_WEDGE:
  ------------------
  |  Branch (1858:13): [True: 3.89k, False: 34.9k]
  ------------------
 1859|  14.7k|            case COMP_INTER_SEG:
  ------------------
  |  Branch (1859:13): [True: 10.8k, False: 27.9k]
  ------------------
 1860|  14.7k|                dsp->mc.mask(uvdst, f->cur.stride[1],
 1861|  14.7k|                             tmp[b->mask_sign], tmp[!b->mask_sign],
 1862|  14.7k|                             bw4 * 4 >> ss_hor, bh4 * 4 >> ss_ver, mask
 1863|  14.7k|                             HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  14.7k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1864|  14.7k|                break;
 1865|  38.8k|            }
 1866|  38.8k|        }
 1867|  51.4k|    }
 1868|       |
 1869|  1.04M|    if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   34|  1.04M|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 1.04M]
  |  |  ------------------
  |  |   35|  1.04M|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  1.04M|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                  if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS) {
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1870|      0|        hex_dump(dst, f->cur.stride[0], b_dim[0] * 4, b_dim[1] * 4, "y-pred");
 1871|      0|        if (has_chroma) {
  ------------------
  |  Branch (1871:13): [True: 0, False: 0]
  ------------------
 1872|      0|            hex_dump(&((pixel *) f->cur.data[1])[uvdstoff], f->cur.stride[1],
 1873|      0|                     cbw4 * 4, cbh4 * 4, "u-pred");
 1874|      0|            hex_dump(&((pixel *) f->cur.data[2])[uvdstoff], f->cur.stride[1],
 1875|      0|                     cbw4 * 4, cbh4 * 4, "v-pred");
 1876|      0|        }
 1877|      0|    }
 1878|       |
 1879|  1.04M|    const int cw4 = (w4 + ss_hor) >> ss_hor, ch4 = (h4 + ss_ver) >> ss_ver;
 1880|       |
 1881|  1.04M|    if (b->skip) {
  ------------------
  |  Branch (1881:9): [True: 876k, False: 163k]
  ------------------
 1882|       |        // reset coef contexts
 1883|   876k|        BlockContext *const a = t->a;
 1884|   876k|        dav1d_memset_pow2[b_dim[2]](&a->lcoef[bx4], 0x40);
 1885|   876k|        dav1d_memset_pow2[b_dim[3]](&t->l.lcoef[by4], 0x40);
 1886|   876k|        if (has_chroma) {
  ------------------
  |  Branch (1886:13): [True: 160k, False: 715k]
  ------------------
 1887|   160k|            dav1d_memset_pow2_fn memset_cw = dav1d_memset_pow2[ulog2(cbw4)];
 1888|   160k|            dav1d_memset_pow2_fn memset_ch = dav1d_memset_pow2[ulog2(cbh4)];
 1889|   160k|            memset_cw(&a->ccoef[0][cbx4], 0x40);
 1890|   160k|            memset_cw(&a->ccoef[1][cbx4], 0x40);
 1891|   160k|            memset_ch(&t->l.ccoef[0][cby4], 0x40);
 1892|   160k|            memset_ch(&t->l.ccoef[1][cby4], 0x40);
 1893|   160k|        }
 1894|   876k|        return 0;
 1895|   876k|    }
 1896|       |
 1897|   163k|    const TxfmInfo *const uvtx = &dav1d_txfm_dimensions[b->uvtx];
 1898|   163k|    const TxfmInfo *const ytx = &dav1d_txfm_dimensions[b->max_ytx];
 1899|   163k|    const uint16_t tx_split[2] = { b->tx_split0, b->tx_split1 };
 1900|       |
 1901|   330k|    for (int init_y = 0; init_y < bh4; init_y += 16) {
  ------------------
  |  Branch (1901:26): [True: 167k, False: 163k]
  ------------------
 1902|   340k|        for (int init_x = 0; init_x < bw4; init_x += 16) {
  ------------------
  |  Branch (1902:30): [True: 173k, False: 167k]
  ------------------
 1903|       |            // coefficient coding & inverse transforms
 1904|   173k|            int y_off = !!init_y, y;
 1905|   173k|            dst += PXSTRIDE(f->cur.stride[0]) * 4 * init_y;
 1906|   353k|            for (y = init_y, t->by += init_y; y < imin(h4, init_y + 16);
  ------------------
  |  Branch (1906:47): [True: 179k, False: 173k]
  ------------------
 1907|   179k|                 y += ytx->h, y_off++)
 1908|   179k|            {
 1909|   179k|                int x, x_off = !!init_x;
 1910|   369k|                for (x = init_x, t->bx += init_x; x < imin(w4, init_x + 16);
  ------------------
  |  Branch (1910:51): [True: 189k, False: 179k]
  ------------------
 1911|   189k|                     x += ytx->w, x_off++)
 1912|   189k|                {
 1913|   189k|                    read_coef_tree(t, bs, b, b->max_ytx, 0, tx_split,
 1914|   189k|                                   x_off, y_off, &dst[x * 4]);
 1915|   189k|                    t->bx += ytx->w;
 1916|   189k|                }
 1917|   179k|                dst += PXSTRIDE(f->cur.stride[0]) * 4 * ytx->h;
 1918|   179k|                t->bx -= x;
 1919|   179k|                t->by += ytx->h;
 1920|   179k|            }
 1921|   173k|            dst -= PXSTRIDE(f->cur.stride[0]) * 4 * y;
 1922|   173k|            t->by -= y;
 1923|       |
 1924|       |            // chroma coefs and inverse transform
 1925|   299k|            if (has_chroma) for (int pl = 0; pl < 2; pl++) {
  ------------------
  |  Branch (1925:17): [True: 99.7k, False: 73.7k]
  |  Branch (1925:46): [True: 199k, False: 99.7k]
  ------------------
 1926|   199k|                pixel *uvdst = ((pixel *) f->cur.data[1 + pl]) + uvdstoff +
 1927|   199k|                    (PXSTRIDE(f->cur.stride[1]) * init_y * 4 >> ss_ver);
 1928|   199k|                for (y = init_y >> ss_ver, t->by += init_y;
 1929|   414k|                     y < imin(ch4, (init_y + 16) >> ss_ver); y += uvtx->h)
  ------------------
  |  Branch (1929:22): [True: 215k, False: 199k]
  ------------------
 1930|   215k|                {
 1931|   215k|                    int x;
 1932|   215k|                    for (x = init_x >> ss_hor, t->bx += init_x;
 1933|   451k|                         x < imin(cw4, (init_x + 16) >> ss_hor); x += uvtx->w)
  ------------------
  |  Branch (1933:26): [True: 236k, False: 215k]
  ------------------
 1934|   236k|                    {
 1935|   236k|                        coef *cf;
 1936|   236k|                        int eob;
 1937|   236k|                        enum TxfmType txtp;
 1938|   236k|                        if (t->frame_thread.pass) {
  ------------------
  |  Branch (1938:29): [True: 236k, False: 18.4E]
  ------------------
 1939|   236k|                            const int p = t->frame_thread.pass & 1;
 1940|   236k|                            const int cbi = *ts->frame_thread[p].cbi++;
 1941|   236k|                            cf = ts->frame_thread[p].cf;
 1942|   236k|                            ts->frame_thread[p].cf += uvtx->w * uvtx->h * 16;
 1943|   236k|                            eob  = cbi >> 5;
 1944|   236k|                            txtp = cbi & 0x1f;
 1945|  18.4E|                        } else {
 1946|  18.4E|                            uint8_t cf_ctx;
 1947|  18.4E|                            cf = bitfn(t->cf);
  ------------------
  |  |   77|  18.4E|#define bitfn(x) x##_16bpc
  ------------------
 1948|  18.4E|                            txtp = t->scratch.txtp_map[(by4 + (y << ss_ver)) * 32 +
 1949|  18.4E|                                                        bx4 + (x << ss_hor)];
 1950|  18.4E|                            eob = decode_coefs(t, &t->a->ccoef[pl][cbx4 + x],
 1951|  18.4E|                                               &t->l.ccoef[pl][cby4 + y],
 1952|  18.4E|                                               b->uvtx, bs, b, 0, 1 + pl,
 1953|  18.4E|                                               cf, &txtp, &cf_ctx);
 1954|  18.4E|                            if (DEBUG_BLOCK_INFO)
  ------------------
  |  |   34|  18.4E|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 18.4E]
  |  |  ------------------
  |  |   35|  18.4E|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  18.4E|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 1955|      0|                                printf("Post-uv-cf-blk[pl=%d,tx=%d,"
 1956|      0|                                       "txtp=%d,eob=%d]: r=%d\n",
 1957|      0|                                       pl, b->uvtx, txtp, eob, ts->msac.rng);
 1958|  18.4E|                            int ctw = imin(uvtx->w, (f->bw - t->bx + ss_hor) >> ss_hor);
 1959|  18.4E|                            int cth = imin(uvtx->h, (f->bh - t->by + ss_ver) >> ss_ver);
 1960|  18.4E|                            dav1d_memset_likely_pow2(&t->a->ccoef[pl][cbx4 + x], cf_ctx, ctw);
 1961|  18.4E|                            dav1d_memset_likely_pow2(&t->l.ccoef[pl][cby4 + y], cf_ctx, cth);
 1962|  18.4E|                        }
 1963|   236k|                        if (eob >= 0) {
  ------------------
  |  Branch (1963:29): [True: 85.5k, False: 151k]
  ------------------
 1964|  85.5k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|  85.5k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 85.5k]
  |  |  ------------------
  |  |   35|  85.5k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  85.5k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1965|      0|                                coef_dump(cf, uvtx->h * 4, uvtx->w * 4, 3, "dq");
 1966|  85.5k|                            dsp->itx.itxfm_add[b->uvtx]
 1967|  85.5k|                                              [txtp](&uvdst[4 * x],
 1968|  85.5k|                                                     f->cur.stride[1],
 1969|  85.5k|                                                     cf, eob HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  85.5k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 1970|  85.5k|                            if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   34|  85.5k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 85.5k]
  |  |  ------------------
  |  |   35|  85.5k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  85.5k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (DEBUG_BLOCK_INFO && DEBUG_B_PIXELS)
  ------------------
  |  |   37|      0|#define DEBUG_B_PIXELS 0
  |  |  ------------------
  |  |  |  Branch (37:24): [Folded, False: 0]
  |  |  ------------------
  ------------------
 1971|      0|                                hex_dump(&uvdst[4 * x], f->cur.stride[1],
 1972|      0|                                         uvtx->w * 4, uvtx->h * 4, "recon");
 1973|  85.5k|                        }
 1974|   236k|                        t->bx += uvtx->w << ss_hor;
 1975|   236k|                    }
 1976|   215k|                    uvdst += PXSTRIDE(f->cur.stride[1]) * 4 * uvtx->h;
 1977|   215k|                    t->bx -= x << ss_hor;
 1978|   215k|                    t->by += uvtx->h << ss_ver;
 1979|   215k|                }
 1980|   199k|                t->by -= y << ss_ver;
 1981|   199k|            }
 1982|   173k|        }
 1983|   167k|    }
 1984|   163k|    return 0;
 1985|  1.04M|}
dav1d_filter_sbrow_deblock_cols_16bpc:
 1987|   702k|void bytefn(dav1d_filter_sbrow_deblock_cols)(Dav1dFrameContext *const f, const int sby) {
 1988|   702k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_DEBLOCK) ||
  ------------------
  |  Branch (1988:9): [True: 8, False: 702k]
  ------------------
 1989|   702k|        (!f->frame_hdr->loopfilter.level_y[0] && !f->frame_hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (1989:10): [True: 8.49k, False: 693k]
  |  Branch (1989:50): [True: 0, False: 8.49k]
  ------------------
 1990|      0|    {
 1991|      0|        return;
 1992|      0|    }
 1993|   702k|    const int y = sby * f->sb_step * 4;
 1994|   702k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 1995|   702k|    pixel *const p[3] = {
 1996|   702k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
 1997|   702k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 1998|   702k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
 1999|   702k|    };
 2000|   702k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2001|   702k|    bytefn(dav1d_loopfilter_sbrow_cols)(f, p, mask, sby,
  ------------------
  |  |   87|   702k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   702k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2002|   702k|                                        f->lf.start_of_tile_row[sby]);
 2003|   702k|}
dav1d_filter_sbrow_deblock_rows_16bpc:
 2005|   932k|void bytefn(dav1d_filter_sbrow_deblock_rows)(Dav1dFrameContext *const f, const int sby) {
 2006|   932k|    const int y = sby * f->sb_step * 4;
 2007|   932k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2008|   932k|    pixel *const p[3] = {
 2009|   932k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
 2010|   932k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2011|   932k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
 2012|   932k|    };
 2013|   932k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2014|   932k|    if (f->c->inloop_filters & DAV1D_INLOOPFILTER_DEBLOCK &&
  ------------------
  |  Branch (2014:9): [True: 931k, False: 574]
  ------------------
 2015|   931k|        (f->frame_hdr->loopfilter.level_y[0] || f->frame_hdr->loopfilter.level_y[1]))
  ------------------
  |  Branch (2015:10): [True: 693k, False: 238k]
  |  Branch (2015:49): [True: 8.48k, False: 230k]
  ------------------
 2016|   702k|    {
 2017|   702k|        bytefn(dav1d_loopfilter_sbrow_rows)(f, p, mask, sby);
  ------------------
  |  |   87|   702k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   702k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2018|   702k|    }
 2019|   932k|    if (f->seq_hdr->cdef || f->lf.restore_planes) {
  ------------------
  |  Branch (2019:9): [True: 859k, False: 72.7k]
  |  Branch (2019:29): [True: 11.5k, False: 61.2k]
  ------------------
 2020|       |        // Store loop filtered pixels required by CDEF / LR
 2021|   871k|        bytefn(dav1d_copy_lpf)(f, p, sby);
  ------------------
  |  |   87|   871k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   871k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2022|   871k|    }
 2023|   932k|}
dav1d_filter_sbrow_cdef_16bpc:
 2025|   859k|void bytefn(dav1d_filter_sbrow_cdef)(Dav1dTaskContext *const tc, const int sby) {
 2026|   859k|    const Dav1dFrameContext *const f = tc->f;
 2027|   859k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_CDEF)) return;
  ------------------
  |  Branch (2027:9): [True: 0, False: 859k]
  ------------------
 2028|   859k|    const int sbsz = f->sb_step;
 2029|   859k|    const int y = sby * sbsz * 4;
 2030|   859k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2031|   859k|    pixel *const p[3] = {
 2032|   859k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
 2033|   859k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2034|   859k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
 2035|   859k|    };
 2036|   859k|    Av1Filter *prev_mask = f->lf.mask + ((sby - 1) >> !f->seq_hdr->sb128) * f->sb128w;
 2037|   859k|    Av1Filter *mask = f->lf.mask + (sby >> !f->seq_hdr->sb128) * f->sb128w;
 2038|   859k|    const int start = sby * sbsz;
 2039|   859k|    if (sby) {
  ------------------
  |  Branch (2039:9): [True: 714k, False: 144k]
  ------------------
 2040|   714k|        const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2041|   714k|        pixel *p_up[3] = {
 2042|   714k|            p[0] - 8 * PXSTRIDE(f->cur.stride[0]),
 2043|   714k|            p[1] - (8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2044|   714k|            p[2] - (8 * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2045|   714k|        };
 2046|   714k|        bytefn(dav1d_cdef_brow)(tc, p_up, prev_mask, start - 2, start, 1, sby);
  ------------------
  |  |   87|   714k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   714k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2047|   714k|    }
 2048|   859k|    const int n_blks = sbsz - 2 * (sby + 1 < f->sbh);
 2049|   859k|    const int end = imin(start + n_blks, f->bh);
 2050|   859k|    bytefn(dav1d_cdef_brow)(tc, p, mask, start, end, 0, sby);
  ------------------
  |  |   87|   859k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   859k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2051|   859k|}
dav1d_filter_sbrow_resize_16bpc:
 2053|  29.8k|void bytefn(dav1d_filter_sbrow_resize)(Dav1dFrameContext *const f, const int sby) {
 2054|  29.8k|    const int sbsz = f->sb_step;
 2055|  29.8k|    const int y = sby * sbsz * 4;
 2056|  29.8k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2057|  29.8k|    const pixel *const p[3] = {
 2058|  29.8k|        f->lf.p[0] + y * PXSTRIDE(f->cur.stride[0]),
 2059|  29.8k|        f->lf.p[1] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver),
 2060|  29.8k|        f->lf.p[2] + (y * PXSTRIDE(f->cur.stride[1]) >> ss_ver)
 2061|  29.8k|    };
 2062|  29.8k|    pixel *const sr_p[3] = {
 2063|  29.8k|        f->lf.sr_p[0] + y * PXSTRIDE(f->sr_cur.p.stride[0]),
 2064|  29.8k|        f->lf.sr_p[1] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver),
 2065|  29.8k|        f->lf.sr_p[2] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver)
 2066|  29.8k|    };
 2067|  29.8k|    const int has_chroma = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400;
 2068|  78.8k|    for (int pl = 0; pl < 1 + 2 * has_chroma; pl++) {
  ------------------
  |  Branch (2068:22): [True: 49.0k, False: 29.8k]
  ------------------
 2069|  49.0k|        const int ss_ver = pl && f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  ------------------
  |  Branch (2069:28): [True: 19.2k, False: 29.8k]
  |  Branch (2069:34): [True: 9.58k, False: 9.61k]
  ------------------
 2070|  49.0k|        const int h_start = 8 * !!sby >> ss_ver;
 2071|  49.0k|        const ptrdiff_t dst_stride = f->sr_cur.p.stride[!!pl];
 2072|  49.0k|        pixel *dst = sr_p[pl] - h_start * PXSTRIDE(dst_stride);
 2073|  49.0k|        const ptrdiff_t src_stride = f->cur.stride[!!pl];
 2074|  49.0k|        const pixel *src = p[pl] - h_start * PXSTRIDE(src_stride);
 2075|  49.0k|        const int h_end = 4 * (sbsz - 2 * (sby + 1 < f->sbh)) >> ss_ver;
 2076|  49.0k|        const int ss_hor = pl && f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
  ------------------
  |  Branch (2076:28): [True: 19.2k, False: 29.8k]
  |  Branch (2076:34): [True: 10.5k, False: 8.61k]
  ------------------
 2077|  49.0k|        const int dst_w = (f->sr_cur.p.p.w + ss_hor) >> ss_hor;
 2078|  49.0k|        const int src_w = (4 * f->bw + ss_hor) >> ss_hor;
 2079|  49.0k|        const int img_h = (f->cur.p.h - sbsz * 4 * sby + ss_ver) >> ss_ver;
 2080|       |
 2081|  49.0k|        f->dsp->mc.resize(dst, dst_stride, src, src_stride, dst_w,
 2082|  49.0k|                          imin(img_h, h_end) + h_start, src_w,
 2083|  49.0k|                          f->resize_step[!!pl], f->resize_start[!!pl]
 2084|  49.0k|                          HIGHBD_CALL_SUFFIX);
  ------------------
  |  |   73|  49.0k|#define HIGHBD_CALL_SUFFIX , f->bitdepth_max
  ------------------
 2085|  49.0k|    }
 2086|  29.8k|}
dav1d_filter_sbrow_lr_16bpc:
 2088|  42.4k|void bytefn(dav1d_filter_sbrow_lr)(Dav1dFrameContext *const f, const int sby) {
 2089|  42.4k|    if (!(f->c->inloop_filters & DAV1D_INLOOPFILTER_RESTORATION)) return;
  ------------------
  |  Branch (2089:9): [True: 0, False: 42.4k]
  ------------------
 2090|  42.4k|    const int y = sby * f->sb_step * 4;
 2091|  42.4k|    const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2092|  42.4k|    pixel *const sr_p[3] = {
 2093|  42.4k|        f->lf.sr_p[0] + y * PXSTRIDE(f->sr_cur.p.stride[0]),
 2094|  42.4k|        f->lf.sr_p[1] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver),
 2095|  42.4k|        f->lf.sr_p[2] + (y * PXSTRIDE(f->sr_cur.p.stride[1]) >> ss_ver)
 2096|  42.4k|    };
 2097|  42.4k|    bytefn(dav1d_lr_sbrow)(f, sr_p, sby);
  ------------------
  |  |   87|  42.4k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  42.4k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2098|  42.4k|}
dav1d_backup_ipred_edge_16bpc:
 2111|  1.00M|void bytefn(dav1d_backup_ipred_edge)(Dav1dTaskContext *const t) {
 2112|  1.00M|    const Dav1dFrameContext *const f = t->f;
 2113|  1.00M|    Dav1dTileState *const ts = t->ts;
 2114|  1.00M|    const int sby = t->by >> f->sb_shift;
 2115|  1.00M|    const int sby_off = f->sb128w * 128 * sby;
 2116|  1.00M|    const int x_off = ts->tiling.col_start;
 2117|       |
 2118|  1.00M|    const pixel *const y =
 2119|  1.00M|        ((const pixel *) f->cur.data[0]) + x_off * 4 +
 2120|  1.00M|                    ((t->by + f->sb_step) * 4 - 1) * PXSTRIDE(f->cur.stride[0]);
 2121|  1.00M|    pixel_copy(&f->ipred_edge[0][sby_off + x_off * 4], y,
  ------------------
  |  |   65|  1.00M|#define pixel_copy(a, b, c) memcpy(a, b, (c) << 1)
  ------------------
 2122|  1.00M|               4 * (ts->tiling.col_end - x_off));
 2123|       |
 2124|  1.00M|    if (f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I400) {
  ------------------
  |  Branch (2124:9): [True: 258k, False: 742k]
  ------------------
 2125|   258k|        const int ss_ver = f->cur.p.layout == DAV1D_PIXEL_LAYOUT_I420;
 2126|   258k|        const int ss_hor = f->cur.p.layout != DAV1D_PIXEL_LAYOUT_I444;
 2127|       |
 2128|   258k|        const ptrdiff_t uv_off = (x_off * 4 >> ss_hor) +
 2129|   258k|            (((t->by + f->sb_step) * 4 >> ss_ver) - 1) * PXSTRIDE(f->cur.stride[1]);
 2130|   776k|        for (int pl = 1; pl <= 2; pl++)
  ------------------
  |  Branch (2130:26): [True: 517k, False: 258k]
  ------------------
 2131|   517k|            pixel_copy(&f->ipred_edge[pl][sby_off + (x_off * 4 >> ss_hor)],
  ------------------
  |  |   65|   517k|#define pixel_copy(a, b, c) memcpy(a, b, (c) << 1)
  ------------------
 2132|   258k|                       &((const pixel *) f->cur.data[pl])[uv_off],
 2133|   258k|                       4 * (ts->tiling.col_end - x_off) >> ss_hor);
 2134|   258k|    }
 2135|  1.00M|}
dav1d_copy_pal_block_y_16bpc:
 2141|  53.5k|{
 2142|  53.5k|    const Dav1dFrameContext *const f = t->f;
 2143|  53.5k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2143:24): [True: 53.5k, False: 1]
  ------------------
 2144|  53.5k|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2145|  53.5k|                            ((t->bx >> 1) + (t->by & 1))][0] :
 2146|  53.5k|        bytefn(t->scratch.pal)[0];
  ------------------
  |  |   87|      1|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  53.5k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2147|   265k|    for (int x = 0; x < bw4; x++)
  ------------------
  |  Branch (2147:21): [True: 212k, False: 53.5k]
  ------------------
 2148|   212k|        memcpy(bytefn(t->al_pal)[0][bx4 + x][0], pal, 8 * sizeof(pixel));
  ------------------
  |  |   87|   212k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   212k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2149|   193k|    for (int y = 0; y < bh4; y++)
  ------------------
  |  Branch (2149:21): [True: 139k, False: 53.5k]
  ------------------
 2150|   139k|        memcpy(bytefn(t->al_pal)[1][by4 + y][0], pal, 8 * sizeof(pixel));
  ------------------
  |  |   87|   139k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   139k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2151|  53.5k|}
dav1d_copy_pal_block_uv_16bpc:
 2157|  13.8k|{
 2158|  13.8k|    const Dav1dFrameContext *const f = t->f;
 2159|  13.8k|    const pixel (*const pal)[8] = t->frame_thread.pass ?
  ------------------
  |  Branch (2159:35): [True: 13.8k, False: 0]
  ------------------
 2160|  13.8k|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2161|  13.8k|                            ((t->bx >> 1) + (t->by & 1))] :
 2162|  13.8k|        bytefn(t->scratch.pal);
  ------------------
  |  |   87|      0|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|      0|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2163|       |    // see aomedia bug 2183 for why we use luma coordinates here
 2164|  41.5k|    for (int pl = 1; pl <= 2; pl++) {
  ------------------
  |  Branch (2164:22): [True: 27.7k, False: 13.8k]
  ------------------
 2165|   147k|        for (int x = 0; x < bw4; x++)
  ------------------
  |  Branch (2165:25): [True: 119k, False: 27.7k]
  ------------------
 2166|   119k|            memcpy(bytefn(t->al_pal)[0][bx4 + x][pl], pal[pl], 8 * sizeof(pixel));
  ------------------
  |  |   87|   119k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   119k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2167|   129k|        for (int y = 0; y < bh4; y++)
  ------------------
  |  Branch (2167:25): [True: 101k, False: 27.7k]
  ------------------
 2168|   101k|            memcpy(bytefn(t->al_pal)[1][by4 + y][pl], pal[pl], 8 * sizeof(pixel));
  ------------------
  |  |   87|   101k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|   101k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2169|  27.7k|    }
 2170|  13.8k|}
dav1d_read_pal_plane_16bpc:
 2175|  67.3k|{
 2176|  67.3k|    Dav1dTileState *const ts = t->ts;
 2177|  67.3k|    const Dav1dFrameContext *const f = t->f;
 2178|  67.3k|    const int pal_sz = b->pal_sz[pl] = dav1d_msac_decode_symbol_adapt8(&ts->msac,
  ------------------
  |  |   48|  67.3k|#define dav1d_msac_decode_symbol_adapt8  dav1d_msac_decode_symbol_adapt8_sse2
  ------------------
 2179|  67.3k|                                           ts->cdf.m.pal_sz[pl][sz_ctx], 6) + 2;
 2180|  67.3k|    pixel cache[16], used_cache[8];
 2181|  67.3k|    int l_cache = pl ? t->pal_sz_uv[1][by4] : t->l.pal_sz[by4];
  ------------------
  |  Branch (2181:19): [True: 13.8k, False: 53.5k]
  ------------------
 2182|  67.3k|    int n_cache = 0;
 2183|       |    // don't reuse above palette outside SB64 boundaries
 2184|  67.3k|    int a_cache = by4 & 15 ? pl ? t->pal_sz_uv[0][bx4] : t->a->pal_sz[bx4] : 0;
  ------------------
  |  Branch (2184:19): [True: 59.9k, False: 7.43k]
  |  Branch (2184:30): [True: 11.7k, False: 48.1k]
  ------------------
 2185|  67.3k|    const pixel *l = bytefn(t->al_pal)[1][by4][pl];
  ------------------
  |  |   87|  67.3k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  67.3k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2186|  67.3k|    const pixel *a = bytefn(t->al_pal)[0][bx4][pl];
  ------------------
  |  |   87|  67.3k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  67.3k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2187|       |
 2188|       |    // fill/sort cache
 2189|   135k|    while (l_cache && a_cache) {
  ------------------
  |  Branch (2189:12): [True: 88.6k, False: 47.0k]
  |  Branch (2189:23): [True: 68.3k, False: 20.3k]
  ------------------
 2190|  68.3k|        if (*l < *a) {
  ------------------
  |  Branch (2190:13): [True: 32.6k, False: 35.6k]
  ------------------
 2191|  32.6k|            if (!n_cache || cache[n_cache - 1] != *l)
  ------------------
  |  Branch (2191:17): [True: 6.43k, False: 26.2k]
  |  Branch (2191:29): [True: 25.3k, False: 929]
  ------------------
 2192|  31.7k|                cache[n_cache++] = *l;
 2193|  32.6k|            l++;
 2194|  32.6k|            l_cache--;
 2195|  35.6k|        } else {
 2196|  35.6k|            if (*a == *l) {
  ------------------
  |  Branch (2196:17): [True: 12.2k, False: 23.4k]
  ------------------
 2197|  12.2k|                l++;
 2198|  12.2k|                l_cache--;
 2199|  12.2k|            }
 2200|  35.6k|            if (!n_cache || cache[n_cache - 1] != *a)
  ------------------
  |  Branch (2200:17): [True: 4.99k, False: 30.6k]
  |  Branch (2200:29): [True: 30.1k, False: 430]
  ------------------
 2201|  35.1k|                cache[n_cache++] = *a;
 2202|  35.6k|            a++;
 2203|  35.6k|            a_cache--;
 2204|  35.6k|        }
 2205|  68.3k|    }
 2206|  67.3k|    if (l_cache) {
  ------------------
  |  Branch (2206:9): [True: 20.3k, False: 46.9k]
  ------------------
 2207|  85.6k|        do {
 2208|  85.6k|            if (!n_cache || cache[n_cache - 1] != *l)
  ------------------
  |  Branch (2208:17): [True: 16.5k, False: 69.0k]
  |  Branch (2208:29): [True: 56.4k, False: 12.6k]
  ------------------
 2209|  72.9k|                cache[n_cache++] = *l;
 2210|  85.6k|            l++;
 2211|  85.6k|        } while (--l_cache > 0);
  ------------------
  |  Branch (2211:18): [True: 65.2k, False: 20.3k]
  ------------------
 2212|  46.9k|    } else if (a_cache) {
  ------------------
  |  Branch (2212:16): [True: 19.1k, False: 27.8k]
  ------------------
 2213|  86.7k|        do {
 2214|  86.7k|            if (!n_cache || cache[n_cache - 1] != *a)
  ------------------
  |  Branch (2214:17): [True: 12.8k, False: 73.9k]
  |  Branch (2214:29): [True: 62.3k, False: 11.5k]
  ------------------
 2215|  75.1k|                cache[n_cache++] = *a;
 2216|  86.7k|            a++;
 2217|  86.7k|        } while (--a_cache > 0);
  ------------------
  |  Branch (2217:18): [True: 67.6k, False: 19.1k]
  ------------------
 2218|  19.1k|    }
 2219|       |
 2220|       |    // find reused cache entries
 2221|  67.3k|    int i = 0;
 2222|   255k|    for (int n = 0; n < n_cache && i < pal_sz; n++)
  ------------------
  |  Branch (2222:21): [True: 197k, False: 58.3k]
  |  Branch (2222:36): [True: 188k, False: 9.03k]
  ------------------
 2223|   188k|        if (dav1d_msac_decode_bool_equi(&ts->msac))
  ------------------
  |  |   53|   188k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2223:13): [True: 86.1k, False: 102k]
  ------------------
 2224|  86.1k|            used_cache[i++] = cache[n];
 2225|  67.3k|    const int n_used_cache = i;
 2226|       |
 2227|       |    // parse new entries
 2228|  67.3k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2228:24): [True: 67.3k, False: 18.4E]
  ------------------
 2229|  67.3k|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2230|  67.3k|                            ((t->bx >> 1) + (t->by & 1))][pl] :
 2231|  18.4E|        bytefn(t->scratch.pal)[pl];
  ------------------
  |  |   87|  18.4E|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  67.3k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2232|  67.3k|    if (i < pal_sz) {
  ------------------
  |  Branch (2232:9): [True: 56.8k, False: 10.5k]
  ------------------
 2233|  56.8k|        const int bpc = BITDEPTH == 8 ? 8 : f->cur.p.bpc;
  ------------------
  |  Branch (2233:25): [Folded, False: 56.8k]
  ------------------
 2234|  56.8k|        int prev = pal[i++] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2235|       |
 2236|  56.8k|        if (i < pal_sz) {
  ------------------
  |  Branch (2236:13): [True: 49.8k, False: 6.98k]
  ------------------
 2237|  49.8k|            int bits = bpc - 3 + dav1d_msac_decode_bools(&ts->msac, 2);
 2238|  49.8k|            const int max = (1 << bpc) - 1;
 2239|       |
 2240|   126k|            do {
 2241|   126k|                const int delta = dav1d_msac_decode_bools(&ts->msac, bits);
 2242|   126k|                prev = pal[i++] = imin(prev + delta + !pl, max);
 2243|   126k|                if (prev + !pl >= max) {
  ------------------
  |  Branch (2243:21): [True: 22.1k, False: 103k]
  ------------------
 2244|  57.0k|                    for (; i < pal_sz; i++)
  ------------------
  |  Branch (2244:28): [True: 34.9k, False: 22.1k]
  ------------------
 2245|  34.9k|                        pal[i] = max;
 2246|  22.1k|                    break;
 2247|  22.1k|                }
 2248|   103k|                bits = imin(bits, 1 + ulog2(max - prev - !pl));
 2249|   103k|            } while (i < pal_sz);
  ------------------
  |  Branch (2249:22): [True: 76.2k, False: 27.7k]
  ------------------
 2250|  49.8k|        }
 2251|       |
 2252|       |        // merge cache+new entries
 2253|  56.8k|        int n = 0, m = n_used_cache;
 2254|   330k|        for (i = 0; i < pal_sz; i++) {
  ------------------
  |  Branch (2254:21): [True: 273k, False: 56.8k]
  ------------------
 2255|   273k|            if (n < n_used_cache && (m >= pal_sz || used_cache[n] <= pal[m])) {
  ------------------
  |  Branch (2255:17): [True: 92.5k, False: 181k]
  |  Branch (2255:38): [True: 20.2k, False: 72.3k]
  |  Branch (2255:53): [True: 35.5k, False: 36.7k]
  ------------------
 2256|  55.8k|                pal[i] = used_cache[n++];
 2257|   217k|            } else {
 2258|   217k|                assert(m < pal_sz);
  ------------------
  |  Branch (2258:17): [True: 217k, False: 18.4E]
  ------------------
 2259|   217k|                pal[i] = pal[m++];
 2260|   217k|            }
 2261|   273k|        }
 2262|  56.8k|    } else {
 2263|  10.5k|        memcpy(pal, used_cache, n_used_cache * sizeof(*used_cache));
 2264|  10.5k|    }
 2265|       |
 2266|  67.4k|    if (DEBUG_BLOCK_INFO) {
  ------------------
  |  |   34|  67.4k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 67.4k]
  |  |  ------------------
  |  |   35|  67.4k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  67.4k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2267|      0|        printf("Post-pal[pl=%d,sz=%d,cache_size=%d,used_cache=%d]: r=%d, cache=",
 2268|      0|               pl, pal_sz, n_cache, n_used_cache, ts->msac.rng);
 2269|      0|        for (int n = 0; n < n_cache; n++)
  ------------------
  |  Branch (2269:25): [True: 0, False: 0]
  ------------------
 2270|      0|            printf("%c%02x", n ? ' ' : '[', cache[n]);
  ------------------
  |  Branch (2270:30): [True: 0, False: 0]
  ------------------
 2271|      0|        printf("%s, pal=", n_cache ? "]" : "[]");
  ------------------
  |  Branch (2271:28): [True: 0, False: 0]
  ------------------
 2272|      0|        for (int n = 0; n < pal_sz; n++)
  ------------------
  |  Branch (2272:25): [True: 0, False: 0]
  ------------------
 2273|      0|            printf("%c%02x", n ? ' ' : '[', pal[n]);
  ------------------
  |  Branch (2273:30): [True: 0, False: 0]
  ------------------
 2274|      0|        printf("]\n");
 2275|      0|    }
 2276|  67.4k|}
dav1d_read_pal_uv_16bpc:
 2280|  13.8k|{
 2281|  13.8k|    bytefn(dav1d_read_pal_plane)(t, b, 1, sz_ctx, bx4, by4);
  ------------------
  |  |   87|  13.8k|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  13.8k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2282|       |
 2283|       |    // V pal coding
 2284|  13.8k|    Dav1dTileState *const ts = t->ts;
 2285|  13.8k|    const Dav1dFrameContext *const f = t->f;
 2286|  13.8k|    pixel *const pal = t->frame_thread.pass ?
  ------------------
  |  Branch (2286:24): [True: 13.8k, False: 18.4E]
  ------------------
 2287|  13.8k|        f->frame_thread.pal[((t->by >> 1) + (t->bx & 1)) * (f->b4_stride >> 1) +
 2288|  13.8k|                            ((t->bx >> 1) + (t->by & 1))][2] :
 2289|  18.4E|        bytefn(t->scratch.pal)[2];
  ------------------
  |  |   87|  18.4E|#define bytefn(x) bitfn(x)
  |  |  ------------------
  |  |  |  |   77|  13.8k|#define bitfn(x) x##_16bpc
  |  |  ------------------
  ------------------
 2290|  13.8k|    const int bpc = BITDEPTH == 8 ? 8 : f->cur.p.bpc;
  ------------------
  |  Branch (2290:21): [Folded, False: 13.8k]
  ------------------
 2291|  13.8k|    if (dav1d_msac_decode_bool_equi(&ts->msac)) {
  ------------------
  |  |   53|  13.8k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2291:9): [True: 7.18k, False: 6.68k]
  ------------------
 2292|  7.18k|        const int bits = bpc - 4 + dav1d_msac_decode_bools(&ts->msac, 2);
 2293|  7.18k|        int prev = pal[0] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2294|  7.18k|        const int max = (1 << bpc) - 1;
 2295|  28.2k|        for (int i = 1; i < b->pal_sz[1]; i++) {
  ------------------
  |  Branch (2295:25): [True: 21.0k, False: 7.18k]
  ------------------
 2296|  21.0k|            int delta = dav1d_msac_decode_bools(&ts->msac, bits);
 2297|  21.0k|            if (delta && dav1d_msac_decode_bool_equi(&ts->msac)) delta = -delta;
  ------------------
  |  |   53|  20.5k|#define dav1d_msac_decode_bool_equi      dav1d_msac_decode_bool_equi_sse2
  ------------------
  |  Branch (2297:17): [True: 20.5k, False: 509]
  |  Branch (2297:26): [True: 10.4k, False: 10.0k]
  ------------------
 2298|  21.0k|            prev = pal[i] = (prev + delta) & max;
 2299|  21.0k|        }
 2300|  7.18k|    } else {
 2301|  30.5k|        for (int i = 0; i < b->pal_sz[1]; i++)
  ------------------
  |  Branch (2301:25): [True: 23.8k, False: 6.68k]
  ------------------
 2302|  23.8k|            pal[i] = dav1d_msac_decode_bools(&ts->msac, bpc);
 2303|  6.68k|    }
 2304|  13.8k|    if (DEBUG_BLOCK_INFO) {
  ------------------
  |  |   34|  13.8k|#define DEBUG_BLOCK_INFO 0 && \
  |  |  ------------------
  |  |  |  Branch (34:26): [Folded, False: 13.8k]
  |  |  ------------------
  |  |   35|  13.8k|        f->frame_hdr->frame_offset == 2 && t->by >= 0 && t->by < 4 && \
  |  |  ------------------
  |  |  |  Branch (35:9): [True: 0, False: 0]
  |  |  |  Branch (35:44): [True: 0, False: 0]
  |  |  |  Branch (35:58): [True: 0, False: 0]
  |  |  ------------------
  |  |   36|  13.8k|        t->bx >= 8 && t->bx < 12
  |  |  ------------------
  |  |  |  Branch (36:9): [True: 0, False: 0]
  |  |  |  Branch (36:23): [True: 0, False: 0]
  |  |  ------------------
  ------------------
 2305|      0|        printf("Post-pal[pl=2]: r=%d ", ts->msac.rng);
 2306|      0|        for (int n = 0; n < b->pal_sz[1]; n++)
  ------------------
  |  Branch (2306:25): [True: 0, False: 0]
  ------------------
 2307|      0|            printf("%c%02x", n ? ' ' : '[', pal[n]);
  ------------------
  |  Branch (2307:30): [True: 0, False: 0]
  ------------------
 2308|      0|        printf("]\n");
 2309|      0|    }
 2310|  13.8k|}

dav1d_ref_create:
   37|   412k|Dav1dRef *dav1d_ref_create(const enum AllocationType type, size_t size) {
   38|   412k|    size = (size + sizeof(void*) - 1) & ~(sizeof(void*) - 1);
   39|       |
   40|   412k|    uint8_t *const data = dav1d_alloc_aligned(type, size + sizeof(Dav1dRef), 64);
  ------------------
  |  |  134|   412k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
   41|   412k|    if (!data) return NULL;
  ------------------
  |  Branch (41:9): [True: 0, False: 412k]
  ------------------
   42|       |
   43|   412k|    Dav1dRef *const res = (Dav1dRef*)(data + size);
   44|   412k|    res->const_data = res->user_data = res->data = data;
   45|   412k|    atomic_init(&res->ref_cnt, 1);
   46|   412k|    res->free_ref = 0;
   47|   412k|    res->free_callback = default_free_callback;
   48|       |
   49|   412k|    return res;
   50|   412k|}
dav1d_ref_create_using_pool:
   56|   789k|Dav1dRef *dav1d_ref_create_using_pool(Dav1dMemPool *const pool, size_t size) {
   57|   789k|    void *const buf = dav1d_mem_pool_pop(pool, size);
   58|   789k|    if (!buf) return NULL;
  ------------------
  |  Branch (58:9): [True: 0, False: 789k]
  ------------------
   59|       |
   60|       |    /* Store Dav1dRef inside the Dav1dMemPoolBuffer alignment padding */
   61|   789k|    assert(sizeof(Dav1dMemPoolBuffer) + sizeof(Dav1dRef) <= 64);
  ------------------
  |  Branch (61:5): [True: 789k, Folded]
  ------------------
   62|   789k|    Dav1dRef *const res = &((Dav1dRef*)buf)[-1];
   63|   789k|    res->data = buf;
   64|   789k|    res->const_data = pool;
   65|   789k|    atomic_init(&res->ref_cnt, 1);
   66|   789k|    res->free_ref = 0;
   67|   789k|    res->free_callback = pool_free_callback;
   68|   789k|    res->user_data = buf;
   69|       |
   70|   789k|    return res;
   71|   789k|}
dav1d_ref_dec:
   73|  50.7M|void dav1d_ref_dec(Dav1dRef **const pref) {
   74|  50.7M|    assert(pref != NULL);
  ------------------
  |  Branch (74:5): [True: 50.7M, False: 18.4E]
  ------------------
   75|       |
   76|  50.7M|    Dav1dRef *const ref = *pref;
   77|  50.7M|    if (!ref) return;
  ------------------
  |  Branch (77:9): [True: 33.5M, False: 17.1M]
  ------------------
   78|       |
   79|  17.1M|    *pref = NULL;
   80|  17.1M|    if (atomic_fetch_sub(&ref->ref_cnt, 1) == 1) {
  ------------------
  |  Branch (80:9): [True: 1.60M, False: 15.5M]
  ------------------
   81|  1.60M|        const int free_ref = ref->free_ref;
   82|  1.60M|        ref->free_callback(ref->const_data, ref->user_data);
   83|  1.60M|        if (free_ref) dav1d_free(ref);
  ------------------
  |  |  135|      0|#define dav1d_free(ptr) free(ptr)
  ------------------
  |  Branch (83:13): [True: 0, False: 1.60M]
  ------------------
   84|  1.60M|    }
   85|  17.1M|}
ref.c:default_free_callback:
   32|   412k|static void default_free_callback(const uint8_t *const data, void *const user_data) {
   33|   412k|    assert(data == user_data);
  ------------------
  |  Branch (33:5): [True: 412k, False: 0]
  ------------------
   34|   412k|    dav1d_free_aligned(user_data);
  ------------------
  |  |  136|   412k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
   35|   412k|}
ref.c:pool_free_callback:
   52|   789k|static void pool_free_callback(const uint8_t *const data, void *const user_data) {
   53|   789k|    dav1d_mem_pool_push((Dav1dMemPool*)data, user_data);
   54|   789k|}

obu.c:dav1d_ref_is_writable:
   73|   417k|static inline int dav1d_ref_is_writable(Dav1dRef *const ref) {
   74|   417k|    return atomic_load(&ref->ref_cnt) == 1 && ref->data;
  ------------------
  |  Branch (74:12): [True: 417k, False: 0]
  |  Branch (74:47): [True: 417k, False: 0]
  ------------------
   75|   417k|}
obu.c:dav1d_ref_init:
   59|    972|{
   60|    972|    ref->data = NULL;
   61|    972|    ref->const_data = ptr;
   62|       |    atomic_init(&ref->ref_cnt, 1);
   63|    972|    ref->free_ref = free_ref;
   64|    972|    ref->free_callback = free_callback;
   65|    972|    ref->user_data = user_data;
   66|    972|    return ref;
   67|    972|}
obu.c:dav1d_ref_inc:
   69|  9.17k|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|  9.17k|}
picture.c:dav1d_ref_inc:
   69|  13.0M|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|  13.0M|}
picture.c:dav1d_ref_init:
   59|   405k|{
   60|   405k|    ref->data = NULL;
   61|   405k|    ref->const_data = ptr;
   62|       |    atomic_init(&ref->ref_cnt, 1);
   63|   405k|    ref->free_ref = free_ref;
   64|   405k|    ref->free_callback = free_callback;
   65|   405k|    ref->user_data = user_data;
   66|   405k|    return ref;
   67|   405k|}
cdf.c:dav1d_ref_inc:
   69|   463k|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|   463k|}
data.c:dav1d_ref_inc:
   69|   771k|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|   771k|}
decode.c:dav1d_ref_inc:
   69|  1.25M|static inline void dav1d_ref_inc(Dav1dRef *const ref) {
   70|       |    atomic_fetch_add_explicit(&ref->ref_cnt, 1, memory_order_relaxed);
   71|  1.25M|}

dav1d_refmvs_find:
  354|  5.27M|{
  355|  5.27M|    const refmvs_frame *const rf = rt->rf;
  356|  5.27M|    const uint8_t *const b_dim = dav1d_block_dimensions[bs];
  357|  5.27M|    const int bw4 = b_dim[0], w4 = imin(imin(bw4, 16), rt->tile_col.end - bx4);
  358|  5.27M|    const int bh4 = b_dim[1], h4 = imin(imin(bh4, 16), rt->tile_row.end - by4);
  359|  5.27M|    mv gmv[2], tgmv[2];
  360|       |
  361|  5.27M|    *cnt = 0;
  362|  5.27M|    assert(ref.ref[0] >=  0 && ref.ref[0] <= 8 &&
  ------------------
  |  Branch (362:5): [True: 5.27M, False: 18.4E]
  |  Branch (362:5): [True: 5.27M, False: 338]
  |  Branch (362:5): [True: 5.27M, False: 18.4E]
  |  Branch (362:5): [True: 5.27M, False: 18.4E]
  ------------------
  363|  5.27M|           ref.ref[1] >= -1 && ref.ref[1] <= 8);
  364|  5.27M|    if (ref.ref[0] > 0) {
  ------------------
  |  Branch (364:9): [True: 3.93M, False: 1.33M]
  ------------------
  365|  3.93M|        tgmv[0] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[0] - 1],
  366|  3.93M|                             bx4, by4, bw4, bh4, rf->frm_hdr);
  367|  3.93M|        gmv[0] = rf->frm_hdr->gmv[ref.ref[0] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
  ------------------
  |  Branch (367:18): [True: 1.29M, False: 2.64M]
  ------------------
  368|  2.64M|                 tgmv[0] : (mv) { .n = INVALID_MV };
  ------------------
  |  |   40|  2.64M|#define INVALID_MV 0x80008000
  ------------------
  369|  3.93M|    } else {
  370|  1.33M|        tgmv[0] = (mv) { .n = 0 };
  371|  1.33M|        gmv[0] = (mv) { .n = INVALID_MV };
  ------------------
  |  |   40|  1.33M|#define INVALID_MV 0x80008000
  ------------------
  372|  1.33M|    }
  373|  5.27M|    if (ref.ref[1] > 0) {
  ------------------
  |  Branch (373:9): [True: 397k, False: 4.87M]
  ------------------
  374|   397k|        tgmv[1] = get_gmv_2d(&rf->frm_hdr->gmv[ref.ref[1] - 1],
  375|   397k|                             bx4, by4, bw4, bh4, rf->frm_hdr);
  376|   397k|        gmv[1] = rf->frm_hdr->gmv[ref.ref[1] - 1].type > DAV1D_WM_TYPE_TRANSLATION ?
  ------------------
  |  Branch (376:18): [True: 88.9k, False: 308k]
  ------------------
  377|   308k|                 tgmv[1] : (mv) { .n = INVALID_MV };
  ------------------
  |  |   40|   308k|#define INVALID_MV 0x80008000
  ------------------
  378|   397k|    }
  379|       |
  380|       |    // top
  381|  5.27M|    int have_newmv = 0, have_col_mvs = 0, have_row_mvs = 0;
  382|  5.27M|    unsigned max_rows = 0, n_rows = ~0;
  383|  5.27M|    const refmvs_block *b_top;
  384|  5.27M|    if (by4 > rt->tile_row.start) {
  ------------------
  |  Branch (384:9): [True: 4.67M, False: 602k]
  ------------------
  385|  4.67M|        max_rows = imin((by4 - rt->tile_row.start + 1) >> 1, 2 + (bh4 > 1));
  386|  4.67M|        b_top = &rt->r[(by4 & 31) - 1 + 5][bx4];
  387|  4.67M|        n_rows = scan_row(mvstack, cnt, ref, gmv, b_top,
  388|  4.67M|                          bw4, w4, max_rows, bw4 >= 16 ? 4 : 1,
  ------------------
  |  Branch (388:46): [True: 1.89M, False: 2.77M]
  ------------------
  389|  4.67M|                          &have_newmv, &have_row_mvs);
  390|  4.67M|    }
  391|       |
  392|       |    // left
  393|  5.27M|    unsigned max_cols = 0, n_cols = ~0U;
  394|  5.27M|    refmvs_block *const *b_left;
  395|  5.27M|    if (bx4 > rt->tile_col.start) {
  ------------------
  |  Branch (395:9): [True: 3.68M, False: 1.59M]
  ------------------
  396|  3.68M|        max_cols = imin((bx4 - rt->tile_col.start + 1) >> 1, 2 + (bw4 > 1));
  397|  3.68M|        b_left = &rt->r[(by4 & 31) + 5];
  398|  3.68M|        n_cols = scan_col(mvstack, cnt, ref, gmv, b_left,
  399|  3.68M|                          bh4, h4, bx4 - 1, max_cols, bh4 >= 16 ? 4 : 1,
  ------------------
  |  Branch (399:55): [True: 815k, False: 2.86M]
  ------------------
  400|  3.68M|                          &have_newmv, &have_col_mvs);
  401|  3.68M|    }
  402|       |
  403|       |    // top/right
  404|  5.27M|    if (n_rows != ~0U && edge_flags & EDGE_I444_TOP_HAS_RIGHT &&
  ------------------
  |  Branch (404:9): [True: 4.65M, False: 617k]
  |  Branch (404:26): [True: 3.36M, False: 1.28M]
  ------------------
  405|  3.36M|        imax(bw4, bh4) <= 16 && bw4 + bx4 < rt->tile_col.end)
  ------------------
  |  Branch (405:9): [True: 2.67M, False: 694k]
  |  Branch (405:33): [True: 1.75M, False: 918k]
  ------------------
  406|  1.75M|    {
  407|  1.75M|        add_spatial_candidate(mvstack, cnt, 4, &b_top[bw4], ref, gmv,
  408|  1.75M|                              &have_newmv, &have_row_mvs);
  409|  1.75M|    }
  410|       |
  411|  5.27M|    const int nearest_match = have_col_mvs + have_row_mvs;
  412|  5.27M|    const int nearest_cnt = *cnt;
  413|  11.4M|    for (int n = 0; n < nearest_cnt; n++)
  ------------------
  |  Branch (413:21): [True: 6.17M, False: 5.27M]
  ------------------
  414|  6.17M|        mvstack[n].weight += 640;
  415|       |
  416|       |    // temporal
  417|  5.27M|    int globalmv_ctx = rf->frm_hdr->use_ref_frame_mvs;
  418|  5.27M|    if (rf->use_ref_frame_mvs) {
  ------------------
  |  Branch (418:9): [True: 805k, False: 4.46M]
  ------------------
  419|   805k|        const ptrdiff_t stride = rf->rp_stride;
  420|   805k|        const int by8 = by4 >> 1, bx8 = bx4 >> 1;
  421|   805k|        const refmvs_temporal_block *const rbi = &rt->rp_proj[(by8 & 15) * stride + bx8];
  422|   805k|        const refmvs_temporal_block *rb = rbi;
  423|   805k|        const int step_h = bw4 >= 16 ? 2 : 1, step_v = bh4 >= 16 ? 2 : 1;
  ------------------
  |  Branch (423:28): [True: 361k, False: 444k]
  |  Branch (423:56): [True: 363k, False: 442k]
  ------------------
  424|   805k|        const int w8 = imin((w4 + 1) >> 1, 8), h8 = imin((h4 + 1) >> 1, 8);
  425|  3.08M|        for (int y = 0; y < h8; y += step_v) {
  ------------------
  |  Branch (425:25): [True: 2.27M, False: 805k]
  ------------------
  426|  9.07M|            for (int x = 0; x < w8; x+= step_h) {
  ------------------
  |  Branch (426:29): [True: 6.79M, False: 2.27M]
  ------------------
  427|  6.79M|                add_temporal_candidate(rf, mvstack, cnt, &rb[x], ref,
  428|  6.79M|                                       !(x | y) ? &globalmv_ctx : NULL, tgmv);
  ------------------
  |  Branch (428:40): [True: 807k, False: 5.98M]
  ------------------
  429|  6.79M|            }
  430|  2.27M|            rb += stride * step_v;
  431|  2.27M|        }
  432|   805k|        if (imin(bw4, bh4) >= 2 && imax(bw4, bh4) < 16) {
  ------------------
  |  Branch (432:13): [True: 724k, False: 81.8k]
  |  Branch (432:36): [True: 355k, False: 368k]
  ------------------
  433|   355k|            const int bh8 = bh4 >> 1, bw8 = bw4 >> 1;
  434|   355k|            rb = &rbi[bh8 * stride];
  435|   355k|            const int has_bottom = by8 + bh8 < imin(rt->tile_row.end >> 1,
  436|   355k|                                                    (by8 & ~7) + 8);
  437|   355k|            if (has_bottom && bx8 - 1 >= imax(rt->tile_col.start >> 1, bx8 & ~7)) {
  ------------------
  |  Branch (437:17): [True: 246k, False: 109k]
  |  Branch (437:31): [True: 185k, False: 60.8k]
  ------------------
  438|   185k|                add_temporal_candidate(rf, mvstack, cnt, &rb[-1], ref,
  439|   185k|                                       NULL, NULL);
  440|   185k|            }
  441|   355k|            if (bx8 + bw8 < imin(rt->tile_col.end >> 1, (bx8 & ~7) + 8)) {
  ------------------
  |  Branch (441:17): [True: 264k, False: 91.1k]
  ------------------
  442|   264k|                if (has_bottom) {
  ------------------
  |  Branch (442:21): [True: 183k, False: 81.0k]
  ------------------
  443|   183k|                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8], ref,
  444|   183k|                                           NULL, NULL);
  445|   183k|                }
  446|   264k|                if (by8 + bh8 - 1 < imin(rt->tile_row.end >> 1, (by8 & ~7) + 8)) {
  ------------------
  |  Branch (446:21): [True: 264k, False: 260]
  ------------------
  447|   264k|                    add_temporal_candidate(rf, mvstack, cnt, &rb[bw8 - stride],
  448|   264k|                                           ref, NULL, NULL);
  449|   264k|                }
  450|   264k|            }
  451|   355k|        }
  452|   805k|    }
  453|  5.27M|    assert(*cnt <= 8);
  ------------------
  |  Branch (453:5): [True: 5.26M, False: 13.9k]
  ------------------
  454|       |
  455|       |    // top/left (which, confusingly, is part of "secondary" references)
  456|  5.26M|    int have_dummy_newmv_match;
  457|  5.26M|    if ((n_rows | n_cols) != ~0U) {
  ------------------
  |  Branch (457:9): [True: 3.10M, False: 2.16M]
  ------------------
  458|  3.10M|        add_spatial_candidate(mvstack, cnt, 4, &b_top[-1], ref, gmv,
  459|  3.10M|                              &have_dummy_newmv_match, &have_row_mvs);
  460|  3.10M|    }
  461|       |
  462|       |    // "secondary" (non-direct neighbour) top & left edges
  463|       |    // what is different about secondary is that everything is now in 8x8 resolution
  464|  15.7M|    for (int n = 2; n <= 3; n++) {
  ------------------
  |  Branch (464:21): [True: 10.4M, False: 5.26M]
  ------------------
  465|  10.4M|        if ((unsigned) n > n_rows && (unsigned) n <= max_rows) {
  ------------------
  |  Branch (465:13): [True: 4.16M, False: 6.33M]
  |  Branch (465:38): [True: 3.17M, False: 991k]
  ------------------
  466|  3.17M|            n_rows += scan_row(mvstack, cnt, ref, gmv,
  467|  3.17M|                               &rt->r[(((by4 & 31) - 2 * n + 1) | 1) + 5][bx4 | 1],
  468|  3.17M|                               bw4, w4, 1 + max_rows - n, bw4 >= 16 ? 4 : 2,
  ------------------
  |  Branch (468:59): [True: 45.5k, False: 3.12M]
  ------------------
  469|  3.17M|                               &have_dummy_newmv_match, &have_row_mvs);
  470|  3.17M|        }
  471|       |
  472|  10.4M|        if ((unsigned) n > n_cols && (unsigned) n <= max_cols) {
  ------------------
  |  Branch (472:13): [True: 4.60M, False: 5.89M]
  |  Branch (472:38): [True: 3.85M, False: 749k]
  ------------------
  473|  3.85M|            n_cols += scan_col(mvstack, cnt, ref, gmv, &rt->r[((by4 & 31) | 1) + 5],
  474|  3.85M|                               bh4, h4, (bx4 - n * 2 + 1) | 1,
  475|  3.85M|                               1 + max_cols - n, bh4 >= 16 ? 4 : 2,
  ------------------
  |  Branch (475:50): [True: 413k, False: 3.44M]
  ------------------
  476|  3.85M|                               &have_dummy_newmv_match, &have_col_mvs);
  477|  3.85M|        }
  478|  10.4M|    }
  479|  5.26M|    assert(*cnt <= 8);
  ------------------
  |  Branch (479:5): [True: 5.26M, False: 18.4E]
  ------------------
  480|       |
  481|  5.26M|    const int ref_match_count = have_col_mvs + have_row_mvs;
  482|       |
  483|       |    // context build-up
  484|  5.26M|    int refmv_ctx, newmv_ctx;
  485|  5.26M|    switch (nearest_match) {
  ------------------
  |  Branch (485:13): [True: 5.27M, False: 18.4E]
  ------------------
  486|   672k|    case 0:
  ------------------
  |  Branch (486:5): [True: 672k, False: 4.59M]
  ------------------
  487|   672k|        refmv_ctx = imin(2, ref_match_count);
  488|   672k|        newmv_ctx = ref_match_count > 0;
  489|   672k|        break;
  490|  2.80M|    case 1:
  ------------------
  |  Branch (490:5): [True: 2.80M, False: 2.45M]
  ------------------
  491|  2.80M|        refmv_ctx = imin(ref_match_count * 3, 4);
  492|  2.80M|        newmv_ctx = 3 - have_newmv;
  493|  2.80M|        break;
  494|  1.79M|    case 2:
  ------------------
  |  Branch (494:5): [True: 1.79M, False: 3.46M]
  ------------------
  495|  1.79M|        refmv_ctx = 5;
  496|  1.79M|        newmv_ctx = 5 - have_newmv;
  497|  1.79M|        break;
  498|  5.26M|    }
  499|       |
  500|       |    // sorting (nearest, then "secondary")
  501|  5.26M|    int len = nearest_cnt;
  502|  10.4M|    while (len) {
  ------------------
  |  Branch (502:12): [True: 5.22M, False: 5.26M]
  ------------------
  503|  5.22M|        int last = 0;
  504|  7.04M|        for (int n = 1; n < len; n++) {
  ------------------
  |  Branch (504:25): [True: 1.82M, False: 5.22M]
  ------------------
  505|  1.82M|            if (mvstack[n - 1].weight < mvstack[n].weight) {
  ------------------
  |  Branch (505:17): [True: 783k, False: 1.03M]
  ------------------
  506|   783k|#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
  507|   783k|                EXCHANGE(mvstack[n - 1], mvstack[n]);
  ------------------
  |  |  506|   783k|#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (506:80): [Folded, False: 783k]
  |  |  ------------------
  ------------------
  508|   783k|                last = n;
  509|   783k|            }
  510|  1.82M|        }
  511|  5.22M|        len = last;
  512|  5.22M|    }
  513|  5.26M|    len = *cnt;
  514|  7.47M|    while (len > nearest_cnt) {
  ------------------
  |  Branch (514:12): [True: 2.20M, False: 5.26M]
  ------------------
  515|  2.20M|        int last = nearest_cnt;
  516|  4.11M|        for (int n = nearest_cnt + 1; n < len; n++) {
  ------------------
  |  Branch (516:39): [True: 1.90M, False: 2.20M]
  ------------------
  517|  1.90M|            if (mvstack[n - 1].weight < mvstack[n].weight) {
  ------------------
  |  Branch (517:17): [True: 670k, False: 1.23M]
  ------------------
  518|   670k|                EXCHANGE(mvstack[n - 1], mvstack[n]);
  ------------------
  |  |  506|   670k|#define EXCHANGE(a, b) do { refmvs_candidate tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (506:80): [Folded, False: 670k]
  |  |  ------------------
  ------------------
  519|   670k|#undef EXCHANGE
  520|   670k|                last = n;
  521|   670k|            }
  522|  1.90M|        }
  523|  2.20M|        len = last;
  524|  2.20M|    }
  525|       |
  526|  5.26M|    if (ref.ref[1] > 0) {
  ------------------
  |  Branch (526:9): [True: 396k, False: 4.86M]
  ------------------
  527|   396k|        if (*cnt < 2) {
  ------------------
  |  Branch (527:13): [True: 249k, False: 146k]
  ------------------
  528|   249k|            const int sign0 = rf->sign_bias[ref.ref[0] - 1];
  529|   249k|            const int sign1 = rf->sign_bias[ref.ref[1] - 1];
  530|   249k|            const int sz4 = imin(w4, h4);
  531|   249k|            refmvs_candidate *const same = &mvstack[*cnt];
  532|   249k|            int same_count[4] = { 0 };
  533|       |
  534|       |            // non-self references in top
  535|   402k|            if (n_rows != ~0U) for (int x = 0; x < sz4;) {
  ------------------
  |  Branch (535:17): [True: 194k, False: 55.4k]
  |  Branch (535:48): [True: 207k, False: 194k]
  ------------------
  536|   207k|                const refmvs_block *const cand_b = &b_top[x];
  537|   207k|                add_compound_extended_candidate(same, same_count, cand_b,
  538|   207k|                                                sign0, sign1, ref, rf->sign_bias);
  539|   207k|                x += dav1d_block_dimensions[cand_b->bs][0];
  540|   207k|            }
  541|       |
  542|       |            // non-self references in left
  543|   521k|            if (n_cols != ~0U) for (int y = 0; y < sz4;) {
  ------------------
  |  Branch (543:17): [True: 237k, False: 12.1k]
  |  Branch (543:48): [True: 284k, False: 237k]
  ------------------
  544|   284k|                const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
  545|   284k|                add_compound_extended_candidate(same, same_count, cand_b,
  546|   284k|                                                sign0, sign1, ref, rf->sign_bias);
  547|   284k|                y += dav1d_block_dimensions[cand_b->bs][1];
  548|   284k|            }
  549|       |
  550|   249k|            refmvs_candidate *const diff = &same[2];
  551|   249k|            const int *const diff_count = &same_count[2];
  552|       |
  553|       |            // merge together
  554|   750k|            for (int n = 0; n < 2; n++) {
  ------------------
  |  Branch (554:29): [True: 500k, False: 249k]
  ------------------
  555|   500k|                int m = same_count[n];
  556|       |
  557|   500k|                if (m >= 2) continue;
  ------------------
  |  Branch (557:21): [True: 147k, False: 353k]
  ------------------
  558|       |
  559|   353k|                const int l = diff_count[n];
  560|   353k|                if (l) {
  ------------------
  |  Branch (560:21): [True: 322k, False: 30.5k]
  ------------------
  561|   322k|                    same[m].mv.mv[n] = diff[0].mv.mv[n];
  562|   322k|                    if (++m == 2) continue;
  ------------------
  |  Branch (562:25): [True: 211k, False: 111k]
  ------------------
  563|   111k|                    if (l == 2) {
  ------------------
  |  Branch (563:25): [True: 90.5k, False: 20.5k]
  ------------------
  564|  90.5k|                        same[1].mv.mv[n] = diff[1].mv.mv[n];
  565|  90.5k|                        continue;
  566|  90.5k|                    }
  567|   111k|                }
  568|  68.3k|                do {
  569|  68.3k|                    same[m].mv.mv[n] = tgmv[n];
  570|  68.3k|                } while (++m < 2);
  ------------------
  |  Branch (570:26): [True: 17.2k, False: 51.1k]
  ------------------
  571|  51.1k|            }
  572|       |
  573|       |            // if the first extended was the same as the non-extended one,
  574|       |            // then replace it with the second extended one
  575|   249k|            int n = *cnt;
  576|   249k|            if (n == 1 && mvstack[0].mv.n == same[0].mv.n)
  ------------------
  |  Branch (576:17): [True: 141k, False: 108k]
  |  Branch (576:27): [True: 95.4k, False: 45.7k]
  ------------------
  577|  95.4k|                mvstack[1].mv = mvstack[2].mv;
  578|   359k|            do {
  579|   359k|                mvstack[n].weight = 2;
  580|   359k|            } while (++n < 2);
  ------------------
  |  Branch (580:22): [True: 109k, False: 249k]
  ------------------
  581|   249k|            *cnt = 2;
  582|   249k|        }
  583|       |
  584|       |        // clamping
  585|   396k|        const int left = -(bx4 + bw4 + 4) * 4 * 8;
  586|   396k|        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
  587|   396k|        const int top = -(by4 + bh4 + 4) * 4 * 8;
  588|   396k|        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
  589|       |
  590|   396k|        const int n_refmvs = *cnt;
  591|   396k|        int n = 0;
  592|   950k|        do {
  593|   950k|            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
  594|   950k|            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
  595|   950k|            mvstack[n].mv.mv[1].x = iclip(mvstack[n].mv.mv[1].x, left, right);
  596|   950k|            mvstack[n].mv.mv[1].y = iclip(mvstack[n].mv.mv[1].y, top, bottom);
  597|   950k|        } while (++n < n_refmvs);
  ------------------
  |  Branch (597:18): [True: 554k, False: 396k]
  ------------------
  598|       |
  599|   396k|        switch (refmv_ctx >> 1) {
  ------------------
  |  Branch (599:17): [True: 398k, False: 18.4E]
  ------------------
  600|   144k|        case 0:
  ------------------
  |  Branch (600:9): [True: 144k, False: 252k]
  ------------------
  601|   144k|            *ctx = imin(newmv_ctx, 1);
  602|   144k|            break;
  603|   148k|        case 1:
  ------------------
  |  Branch (603:9): [True: 148k, False: 248k]
  ------------------
  604|   148k|            *ctx = 1 + imin(newmv_ctx, 3);
  605|   148k|            break;
  606|   105k|        case 2:
  ------------------
  |  Branch (606:9): [True: 105k, False: 290k]
  ------------------
  607|   105k|            *ctx = iclip(3 + newmv_ctx, 4, 7);
  608|   105k|            break;
  609|   396k|        }
  610|       |
  611|   397k|        return;
  612|  4.86M|    } else if (*cnt < 2 && ref.ref[0] > 0) {
  ------------------
  |  Branch (612:16): [True: 3.08M, False: 1.78M]
  |  Branch (612:28): [True: 2.50M, False: 572k]
  ------------------
  613|  2.50M|        const int sign = rf->sign_bias[ref.ref[0] - 1];
  614|  2.50M|        const int sz4 = imin(w4, h4);
  615|       |
  616|       |        // non-self references in top
  617|  4.73M|        if (n_rows != ~0U) for (int x = 0; x < sz4 && *cnt < 2;) {
  ------------------
  |  Branch (617:13): [True: 2.34M, False: 160k]
  |  Branch (617:44): [True: 2.39M, False: 2.34M]
  |  Branch (617:55): [True: 2.38M, False: 6.00k]
  ------------------
  618|  2.38M|            const refmvs_block *const cand_b = &b_top[x];
  619|  2.38M|            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
  620|  2.38M|            x += dav1d_block_dimensions[cand_b->bs][0];
  621|  2.38M|        }
  622|       |
  623|       |        // non-self references in left
  624|  2.50M|        if (n_cols != ~0U) for (int y = 0; y < sz4 && *cnt < 2;) {
  ------------------
  |  Branch (624:13): [True: 967k, False: 1.54M]
  |  Branch (624:44): [True: 1.01M, False: 835k]
  |  Branch (624:55): [True: 881k, False: 132k]
  ------------------
  625|   881k|            const refmvs_block *const cand_b = &b_left[y][bx4 - 1];
  626|   881k|            add_single_extended_candidate(mvstack, cnt, cand_b, sign, rf->sign_bias);
  627|   881k|            y += dav1d_block_dimensions[cand_b->bs][1];
  628|   881k|        }
  629|  2.50M|    }
  630|  5.26M|    assert(*cnt <= 8);
  ------------------
  |  Branch (630:5): [True: 4.87M, False: 18.4E]
  ------------------
  631|       |
  632|       |    // clamping
  633|  4.87M|    int n_refmvs = *cnt;
  634|  4.87M|    if (n_refmvs) {
  ------------------
  |  Branch (634:9): [True: 4.58M, False: 292k]
  ------------------
  635|  4.58M|        const int left = -(bx4 + bw4 + 4) * 4 * 8;
  636|  4.58M|        const int right = (rf->iw4 - bx4 + 4) * 4 * 8;
  637|  4.58M|        const int top = -(by4 + bh4 + 4) * 4 * 8;
  638|  4.58M|        const int bottom = (rf->ih4 - by4 + 4) * 4 * 8;
  639|       |
  640|  4.58M|        int n = 0;
  641|  9.12M|        do {
  642|  9.12M|            mvstack[n].mv.mv[0].x = iclip(mvstack[n].mv.mv[0].x, left, right);
  643|  9.12M|            mvstack[n].mv.mv[0].y = iclip(mvstack[n].mv.mv[0].y, top, bottom);
  644|  9.12M|        } while (++n < n_refmvs);
  ------------------
  |  Branch (644:18): [True: 4.54M, False: 4.58M]
  ------------------
  645|  4.58M|    }
  646|       |
  647|  7.96M|    for (int n = *cnt; n < 2; n++)
  ------------------
  |  Branch (647:24): [True: 3.09M, False: 4.87M]
  ------------------
  648|  3.09M|        mvstack[n].mv.mv[0] = tgmv[0];
  649|       |
  650|  4.87M|    *ctx = (refmv_ctx << 4) | (globalmv_ctx << 3) | newmv_ctx;
  651|  4.87M|}
dav1d_refmvs_tile_sbrow_init:
  657|  4.01M|{
  658|  4.01M|    if (rf->n_tile_threads == 1) tile_row_idx = 0;
  ------------------
  |  Branch (658:9): [True: 0, False: 4.01M]
  ------------------
  659|  4.01M|    rt->rp_proj = &rf->rp_proj[16 * rf->rp_stride * tile_row_idx];
  660|  4.01M|    const ptrdiff_t r_stride = rf->rp_stride * 2;
  661|  4.01M|    const ptrdiff_t pass_off = (rf->n_frame_threads > 1 && pass == 2) ?
  ------------------
  |  Branch (661:33): [True: 4.01M, False: 44]
  |  Branch (661:60): [True: 1.94M, False: 2.07M]
  ------------------
  662|  2.07M|        35 * 2 * rf->n_blocks : 0;
  663|  4.01M|    refmvs_block *r = &rf->r[35 * r_stride * tile_row_idx + pass_off];
  664|  4.01M|    const int sbsz = rf->sbsz;
  665|  4.01M|    const int off = (sbsz * sby) & 16;
  666|   100M|    for (int i = 0; i < sbsz; i++, r += r_stride)
  ------------------
  |  Branch (666:21): [True: 96.8M, False: 4.01M]
  ------------------
  667|  96.8M|        rt->r[off + 5 + i] = r;
  668|  4.01M|    rt->r[off + 0] = r;
  669|  4.01M|    r += r_stride;
  670|  4.01M|    rt->r[off + 1] = NULL;
  671|  4.01M|    rt->r[off + 2] = r;
  672|  4.01M|    r += r_stride;
  673|  4.01M|    rt->r[off + 3] = NULL;
  674|  4.01M|    rt->r[off + 4] = r;
  675|  4.01M|    if (sby & 1) {
  ------------------
  |  Branch (675:9): [True: 1.80M, False: 2.20M]
  ------------------
  676|  1.80M|#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
  677|  1.80M|        EXCHANGE(rt->r[off + 0], rt->r[off + sbsz + 0]);
  ------------------
  |  |  676|  1.80M|#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (676:75): [Folded, False: 1.80M]
  |  |  ------------------
  ------------------
  678|  1.80M|        EXCHANGE(rt->r[off + 2], rt->r[off + sbsz + 2]);
  ------------------
  |  |  676|  1.80M|#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (676:75): [Folded, False: 1.80M]
  |  |  ------------------
  ------------------
  679|  1.80M|        EXCHANGE(rt->r[off + 4], rt->r[off + sbsz + 4]);
  ------------------
  |  |  676|  1.80M|#define EXCHANGE(a, b) do { void *const tmp = a; a = b; b = tmp; } while (0)
  |  |  ------------------
  |  |  |  Branch (676:75): [Folded, False: 1.80M]
  |  |  ------------------
  ------------------
  680|  1.80M|#undef EXCHANGE
  681|  1.80M|    }
  682|       |
  683|  4.01M|    rt->rf = rf;
  684|  4.01M|    rt->tile_row.start = tile_row_start4;
  685|  4.01M|    rt->tile_row.end = imin(tile_row_end4, rf->ih4);
  686|  4.01M|    rt->tile_col.start = tile_col_start4;
  687|  4.01M|    rt->tile_col.end = imin(tile_col_end4, rf->iw4);
  688|  4.01M|}
dav1d_refmvs_init_frame:
  812|   325k|{
  813|   325k|    const int rp_stride = ((frm_hdr->width[0] + 127) & ~127) >> 3;
  814|  18.4E|    const int n_tile_rows = n_tile_threads > 1 ? frm_hdr->tiling.rows : 1;
  ------------------
  |  Branch (814:29): [True: 325k, False: 18.4E]
  ------------------
  815|   325k|    const int n_blocks = rp_stride * n_tile_rows;
  816|       |
  817|   325k|    rf->sbsz = 16 << seq_hdr->sb128;
  818|   325k|    rf->frm_hdr = frm_hdr;
  819|   325k|    rf->iw8 = (frm_hdr->width[0] + 7) >> 3;
  820|   325k|    rf->ih8 = (frm_hdr->height + 7) >> 3;
  821|   325k|    rf->iw4 = rf->iw8 << 1;
  822|   325k|    rf->ih4 = rf->ih8 << 1;
  823|   325k|    rf->rp = rp;
  824|   325k|    rf->rp_stride = rp_stride;
  825|   325k|    rf->n_tile_threads = n_tile_threads;
  826|   325k|    rf->n_frame_threads = n_frame_threads;
  827|       |
  828|   325k|    if (n_blocks != rf->n_blocks) {
  ------------------
  |  Branch (828:9): [True: 30.5k, False: 295k]
  ------------------
  829|  30.5k|        const size_t r_sz = sizeof(*rf->r) * 35 * 2 * n_blocks * (1 + (n_frame_threads > 1));
  830|  30.5k|        const size_t rp_proj_sz = sizeof(*rf->rp_proj) * 16 * n_blocks;
  831|       |        /* Note that sizeof(*rf->r) == 12, but it's accessed using 16-byte unaligned
  832|       |         * loads in save_tmvs() asm which can overread 4 bytes into rp_proj. */
  833|  30.5k|        dav1d_free_aligned(rf->r);
  ------------------
  |  |  136|  30.5k|#define dav1d_free_aligned(ptr) dav1d_free_aligned_internal(ptr)
  ------------------
  834|  30.5k|        rf->r = dav1d_alloc_aligned(ALLOC_REFMVS, r_sz + rp_proj_sz, 64);
  ------------------
  |  |  134|  30.5k|#define dav1d_alloc_aligned(type, sz, align) dav1d_alloc_aligned_internal(sz, align)
  ------------------
  835|  30.5k|        if (!rf->r) {
  ------------------
  |  Branch (835:13): [True: 0, False: 30.5k]
  ------------------
  836|      0|            rf->n_blocks = 0;
  837|      0|            return DAV1D_ERR(ENOMEM);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  838|      0|        }
  839|       |
  840|  30.5k|        rf->rp_proj = (refmvs_temporal_block*)((uintptr_t)rf->r + r_sz);
  841|  30.5k|        rf->n_blocks = n_blocks;
  842|  30.5k|    }
  843|       |
  844|   325k|    const int poc = frm_hdr->frame_offset;
  845|  2.60M|    for (int i = 0; i < 7; i++) {
  ------------------
  |  Branch (845:21): [True: 2.27M, False: 325k]
  ------------------
  846|  2.27M|        const int poc_diff = get_poc_diff(seq_hdr->order_hint_n_bits,
  847|  2.27M|                                          ref_poc[i], poc);
  848|  2.27M|        rf->sign_bias[i] = poc_diff > 0;
  849|  2.27M|        rf->mfmv_sign[i] = poc_diff < 0;
  850|  2.27M|        rf->pocdiff[i] = iclip(get_poc_diff(seq_hdr->order_hint_n_bits,
  851|  2.27M|                                            poc, ref_poc[i]), -31, 31);
  852|  2.27M|    }
  853|       |
  854|       |    // temporal MV setup
  855|   325k|    rf->n_mfmvs = 0;
  856|   325k|    rf->rp_ref = rp_ref;
  857|   325k|    if (frm_hdr->use_ref_frame_mvs && seq_hdr->order_hint_n_bits) {
  ------------------
  |  Branch (857:9): [True: 75.4k, False: 250k]
  |  Branch (857:39): [True: 75.4k, False: 0]
  ------------------
  858|  75.4k|        int total = 2;
  859|  75.4k|        if (rp_ref[0] && ref_ref_poc[0][6] != ref_poc[3] /* alt-of-last != gold */) {
  ------------------
  |  Branch (859:13): [True: 63.0k, False: 12.4k]
  |  Branch (859:26): [True: 43.2k, False: 19.7k]
  ------------------
  860|  43.2k|            rf->mfmv_ref[rf->n_mfmvs++] = 0; // last
  861|  43.2k|            total = 3;
  862|  43.2k|        }
  863|  75.4k|        if (rp_ref[4] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[4],
  ------------------
  |  Branch (863:13): [True: 63.2k, False: 12.2k]
  |  Branch (863:26): [True: 28.6k, False: 34.5k]
  ------------------
  864|  63.2k|                                      frm_hdr->frame_offset) > 0)
  865|  28.6k|        {
  866|  28.6k|            rf->mfmv_ref[rf->n_mfmvs++] = 4; // bwd
  867|  28.6k|        }
  868|  75.4k|        if (rp_ref[5] && get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[5],
  ------------------
  |  Branch (868:13): [True: 60.1k, False: 15.3k]
  |  Branch (868:26): [True: 26.8k, False: 33.3k]
  ------------------
  869|  60.1k|                                      frm_hdr->frame_offset) > 0)
  870|  26.8k|        {
  871|  26.8k|            rf->mfmv_ref[rf->n_mfmvs++] = 5; // altref2
  872|  26.8k|        }
  873|  75.4k|        if (rf->n_mfmvs < total && rp_ref[6] &&
  ------------------
  |  Branch (873:13): [True: 53.0k, False: 22.4k]
  |  Branch (873:36): [True: 37.1k, False: 15.8k]
  ------------------
  874|  37.1k|            get_poc_diff(seq_hdr->order_hint_n_bits, ref_poc[6],
  ------------------
  |  Branch (874:13): [True: 17.8k, False: 19.2k]
  ------------------
  875|  37.1k|                         frm_hdr->frame_offset) > 0)
  876|  17.8k|        {
  877|  17.8k|            rf->mfmv_ref[rf->n_mfmvs++] = 6; // altref
  878|  17.8k|        }
  879|  75.4k|        if (rf->n_mfmvs < total && rp_ref[1])
  ------------------
  |  Branch (879:13): [True: 42.8k, False: 32.6k]
  |  Branch (879:36): [True: 31.3k, False: 11.5k]
  ------------------
  880|  31.3k|            rf->mfmv_ref[rf->n_mfmvs++] = 1; // last2
  881|       |
  882|   223k|        for (int n = 0; n < rf->n_mfmvs; n++) {
  ------------------
  |  Branch (882:25): [True: 147k, False: 75.4k]
  ------------------
  883|   147k|            const int rpoc = ref_poc[rf->mfmv_ref[n]];
  884|   147k|            const int diff1 = get_poc_diff(seq_hdr->order_hint_n_bits,
  885|   147k|                                           rpoc, frm_hdr->frame_offset);
  886|   147k|            if (abs(diff1) > 31) {
  ------------------
  |  Branch (886:17): [True: 695, False: 147k]
  ------------------
  887|    695|                rf->mfmv_ref2cur[n] = INVALID_REF2CUR;
  ------------------
  |  |   41|    695|#define INVALID_REF2CUR (-32)
  ------------------
  888|   147k|            } else {
  889|   147k|                rf->mfmv_ref2cur[n] = rf->mfmv_ref[n] < 4 ? -diff1 : diff1;
  ------------------
  |  Branch (889:39): [True: 74.1k, False: 73.0k]
  ------------------
  890|  1.17M|                for (int m = 0; m < 7; m++) {
  ------------------
  |  Branch (890:33): [True: 1.02M, False: 147k]
  ------------------
  891|  1.02M|                    const int rrpoc = ref_ref_poc[rf->mfmv_ref[n]][m];
  892|  1.02M|                    const int diff2 = get_poc_diff(seq_hdr->order_hint_n_bits,
  893|  1.02M|                                                   rpoc, rrpoc);
  894|       |                    // unsigned comparison also catches the < 0 case
  895|  1.02M|                    rf->mfmv_ref2ref[n][m] = (unsigned) diff2 > 31U ? 0 : diff2;
  ------------------
  |  Branch (895:46): [True: 260k, False: 769k]
  ------------------
  896|  1.02M|                }
  897|   147k|            }
  898|   147k|        }
  899|  75.4k|    }
  900|   325k|    rf->use_ref_frame_mvs = rf->n_mfmvs > 0;
  901|       |
  902|   325k|    return 0;
  903|   325k|}
dav1d_refmvs_dsp_init:
  926|  9.41k|{
  927|  9.41k|    c->load_tmvs = load_tmvs_c;
  928|  9.41k|    c->save_tmvs = save_tmvs_c;
  929|  9.41k|    c->splat_mv = splat_mv_c;
  930|       |
  931|  9.41k|#if HAVE_ASM
  932|       |#if ARCH_AARCH64 || ARCH_ARM
  933|       |    refmvs_dsp_init_arm(c);
  934|       |#elif ARCH_LOONGARCH64
  935|       |    refmvs_dsp_init_loongarch(c);
  936|       |#elif ARCH_X86
  937|       |    refmvs_dsp_init_x86(c);
  938|  9.41k|#endif
  939|  9.41k|#endif
  940|  9.41k|}
refmvs.c:scan_row:
  102|  7.83M|{
  103|  7.83M|    const refmvs_block *cand_b = b;
  104|  7.83M|    const enum BlockSize first_cand_bs = cand_b->bs;
  105|  7.83M|    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
  106|  7.83M|    int cand_bw4 = first_cand_b_dim[0];
  107|  7.83M|    int len = imax(step, imin(bw4, cand_bw4));
  108|       |
  109|  7.83M|    if (bw4 <= cand_bw4) {
  ------------------
  |  Branch (109:9): [True: 6.96M, False: 869k]
  ------------------
  110|       |        // FIXME weight can be higher for odd blocks (bx4 & 1), but then the
  111|       |        // position of the first block has to be odd already, i.e. not just
  112|       |        // for row_offset=-3/-5
  113|       |        // FIXME why can this not be cand_bw4?
  114|  6.96M|        const int weight = bw4 == 1 ? 2 :
  ------------------
  |  Branch (114:28): [True: 1.53M, False: 5.43M]
  ------------------
  115|  6.96M|                           imax(2, imin(2 * max_rows, first_cand_b_dim[1]));
  116|  6.96M|        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
  117|  6.96M|                              have_newmv_match, have_refmv_match);
  118|  6.96M|        return weight >> 1;
  119|  6.96M|    }
  120|       |
  121|  1.77M|    for (int x = 0;;) {
  122|       |        // FIXME if we overhang above, we could fill a bitmask so we don't have
  123|       |        // to repeat the add_spatial_candidate() for the next row, but just increase
  124|       |        // the weight here
  125|  1.77M|        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
  126|  1.77M|                              have_newmv_match, have_refmv_match);
  127|  1.77M|        x += len;
  128|  1.77M|        if (x >= w4) return 1;
  ------------------
  |  Branch (128:13): [True: 862k, False: 912k]
  ------------------
  129|   912k|        cand_b = &b[x];
  130|   912k|        cand_bw4 = dav1d_block_dimensions[cand_b->bs][0];
  131|   912k|        assert(cand_bw4 < bw4);
  ------------------
  |  Branch (131:9): [True: 914k, False: 18.4E]
  ------------------
  132|   914k|        len = imax(step, cand_bw4);
  133|   914k|    }
  134|   869k|}
refmvs.c:scan_col:
  141|  7.50M|{
  142|  7.50M|    const refmvs_block *cand_b = &b[0][bx4];
  143|  7.50M|    const enum BlockSize first_cand_bs = cand_b->bs;
  144|  7.50M|    const uint8_t *const first_cand_b_dim = dav1d_block_dimensions[first_cand_bs];
  145|  7.50M|    int cand_bh4 = first_cand_b_dim[1];
  146|  7.50M|    int len = imax(step, imin(bh4, cand_bh4));
  147|       |
  148|  7.50M|    if (bh4 <= cand_bh4) {
  ------------------
  |  Branch (148:9): [True: 6.25M, False: 1.25M]
  ------------------
  149|       |        // FIXME weight can be higher for odd blocks (by4 & 1), but then the
  150|       |        // position of the first block has to be odd already, i.e. not just
  151|       |        // for col_offset=-3/-5
  152|       |        // FIXME why can this not be cand_bh4?
  153|  6.25M|        const int weight = bh4 == 1 ? 2 :
  ------------------
  |  Branch (153:28): [True: 1.89M, False: 4.36M]
  ------------------
  154|  6.25M|                           imax(2, imin(2 * max_cols, first_cand_b_dim[0]));
  155|  6.25M|        add_spatial_candidate(mvstack, cnt, len * weight, cand_b, ref, gmv,
  156|  6.25M|                            have_newmv_match, have_refmv_match);
  157|  6.25M|        return weight >> 1;
  158|  6.25M|    }
  159|       |
  160|  2.40M|    for (int y = 0;;) {
  161|       |        // FIXME if we overhang above, we could fill a bitmask so we don't have
  162|       |        // to repeat the add_spatial_candidate() for the next row, but just increase
  163|       |        // the weight here
  164|  2.40M|        add_spatial_candidate(mvstack, cnt, len * 2, cand_b, ref, gmv,
  165|  2.40M|                              have_newmv_match, have_refmv_match);
  166|  2.40M|        y += len;
  167|  2.40M|        if (y >= h4) return 1;
  ------------------
  |  Branch (167:13): [True: 1.26M, False: 1.14M]
  ------------------
  168|  1.14M|        cand_b = &b[y][bx4];
  169|  1.14M|        cand_bh4 = dav1d_block_dimensions[cand_b->bs][1];
  170|  1.14M|        assert(cand_bh4 < bh4);
  ------------------
  |  Branch (170:9): [True: 1.14M, False: 18.4E]
  ------------------
  171|  1.14M|        len = imax(step, cand_bh4);
  172|  1.14M|    }
  173|  1.25M|}
refmvs.c:add_spatial_candidate:
   46|  21.9M|{
   47|  21.9M|    if (b->mv.mv[0].n == INVALID_MV) return; // intra block, no intrabc
  ------------------
  |  |   40|  21.9M|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (47:9): [True: 4.28M, False: 17.6M]
  ------------------
   48|       |
   49|  17.6M|    if (ref.ref[1] == -1) {
  ------------------
  |  Branch (49:9): [True: 15.7M, False: 1.89M]
  ------------------
   50|  19.8M|        for (int n = 0; n < 2; n++) {
  ------------------
  |  Branch (50:25): [True: 17.9M, False: 1.89M]
  ------------------
   51|  17.9M|            if (b->ref.ref[n] == ref.ref[0]) {
  ------------------
  |  Branch (51:17): [True: 13.8M, False: 4.09M]
  ------------------
   52|  13.8M|                const mv cand_mv = ((b->mf & 1) && gmv[0].n != INVALID_MV) ?
  ------------------
  |  |   40|  3.88M|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (52:37): [True: 3.88M, False: 9.98M]
  |  Branch (52:52): [True: 1.34M, False: 2.53M]
  ------------------
   53|  12.5M|                                   gmv[0] : b->mv.mv[n];
   54|       |
   55|  13.8M|                *have_refmv_match = 1;
   56|  13.8M|                *have_newmv_match |= b->mf >> 1;
   57|       |
   58|  13.8M|                const int last = *cnt;
   59|  23.7M|                for (int m = 0; m < last; m++)
  ------------------
  |  Branch (59:33): [True: 15.5M, False: 8.16M]
  ------------------
   60|  15.5M|                    if (mvstack[m].mv.mv[0].n == cand_mv.n) {
  ------------------
  |  Branch (60:25): [True: 5.70M, False: 9.83M]
  ------------------
   61|  5.70M|                        mvstack[m].weight += weight;
   62|  5.70M|                        return;
   63|  5.70M|                    }
   64|       |
   65|  8.21M|                if (last < 8) {
  ------------------
  |  Branch (65:21): [True: 8.21M, False: 18.4E]
  ------------------
   66|  8.21M|                    mvstack[last].mv.mv[0] = cand_mv;
   67|  8.21M|                    mvstack[last].weight = weight;
   68|  8.21M|                    *cnt = last + 1;
   69|  8.21M|                }
   70|  8.16M|                return;
   71|  13.8M|            }
   72|  17.9M|        }
   73|  15.7M|    } else if (b->ref.pair == ref.pair) {
  ------------------
  |  Branch (73:16): [True: 728k, False: 1.16M]
  ------------------
   74|   728k|        const refmvs_mvpair cand_mv = { .mv = {
   75|   728k|            [0] = ((b->mf & 1) && gmv[0].n != INVALID_MV) ? gmv[0] : b->mv.mv[0],
  ------------------
  |  |   40|  55.0k|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (75:20): [True: 55.0k, False: 672k]
  |  Branch (75:35): [True: 29.9k, False: 25.1k]
  ------------------
   76|   728k|            [1] = ((b->mf & 1) && gmv[1].n != INVALID_MV) ? gmv[1] : b->mv.mv[1],
  ------------------
  |  |   40|  55.0k|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (76:20): [True: 55.0k, False: 672k]
  |  Branch (76:35): [True: 20.6k, False: 34.3k]
  ------------------
   77|   728k|        }};
   78|       |
   79|   728k|        *have_refmv_match = 1;
   80|   728k|        *have_newmv_match |= b->mf >> 1;
   81|       |
   82|   728k|        const int last = *cnt;
   83|  1.10M|        for (int n = 0; n < last; n++)
  ------------------
  |  Branch (83:25): [True: 682k, False: 426k]
  ------------------
   84|   682k|            if (mvstack[n].mv.n == cand_mv.n) {
  ------------------
  |  Branch (84:17): [True: 301k, False: 381k]
  ------------------
   85|   301k|                mvstack[n].weight += weight;
   86|   301k|                return;
   87|   301k|            }
   88|       |
   89|   426k|        if (last < 8) {
  ------------------
  |  Branch (89:13): [True: 425k, False: 712]
  ------------------
   90|   425k|            mvstack[last].mv = cand_mv;
   91|   425k|            mvstack[last].weight = weight;
   92|   425k|            *cnt = last + 1;
   93|   425k|        }
   94|   426k|    }
   95|  17.6M|}
refmvs.c:add_temporal_candidate:
  198|  7.38M|{
  199|  7.38M|    if (rb->mv.n == INVALID_MV) return;
  ------------------
  |  |   40|  7.38M|#define INVALID_MV 0x80008000
  ------------------
  |  Branch (199:9): [True: 5.13M, False: 2.25M]
  ------------------
  200|       |
  201|  2.25M|    union mv mv = mv_projection(rb->mv, rf->pocdiff[ref.ref[0] - 1], rb->ref);
  202|  2.25M|    fix_mv_precision(rf->frm_hdr, &mv);
  203|       |
  204|  2.25M|    const int last = *cnt;
  205|  2.25M|    if (ref.ref[1] == -1) {
  ------------------
  |  Branch (205:9): [True: 1.59M, False: 664k]
  ------------------
  206|  1.59M|        if (globalmv_ctx)
  ------------------
  |  Branch (206:13): [True: 312k, False: 1.28M]
  ------------------
  207|   312k|            *globalmv_ctx = (abs(mv.x - gmv[0].x) | abs(mv.y - gmv[0].y)) >= 16;
  208|       |
  209|  4.30M|        for (int n = 0; n < last; n++)
  ------------------
  |  Branch (209:25): [True: 3.83M, False: 476k]
  ------------------
  210|  3.83M|            if (mvstack[n].mv.mv[0].n == mv.n) {
  ------------------
  |  Branch (210:17): [True: 1.11M, False: 2.71M]
  ------------------
  211|  1.11M|                mvstack[n].weight += 2;
  212|  1.11M|                return;
  213|  1.11M|            }
  214|   494k|        if (last < 8) {
  ------------------
  |  Branch (214:13): [True: 494k, False: 18.4E]
  ------------------
  215|   494k|            mvstack[last].mv.mv[0] = mv;
  216|   494k|            mvstack[last].weight = 2;
  217|   494k|            *cnt = last + 1;
  218|   494k|        }
  219|   664k|    } else {
  220|   664k|        refmvs_mvpair mvp = { .mv = {
  221|   664k|            [0] = mv,
  222|   664k|            [1] = mv_projection(rb->mv, rf->pocdiff[ref.ref[1] - 1], rb->ref),
  223|   664k|        }};
  224|   664k|        fix_mv_precision(rf->frm_hdr, &mvp.mv[1]);
  225|       |
  226|  1.51M|        for (int n = 0; n < last; n++)
  ------------------
  |  Branch (226:25): [True: 1.36M, False: 149k]
  ------------------
  227|  1.36M|            if (mvstack[n].mv.n == mvp.n) {
  ------------------
  |  Branch (227:17): [True: 514k, False: 847k]
  ------------------
  228|   514k|                mvstack[n].weight += 2;
  229|   514k|                return;
  230|   514k|            }
  231|   164k|        if (last < 8) {
  ------------------
  |  Branch (231:13): [True: 164k, False: 18.4E]
  ------------------
  232|   164k|            mvstack[last].mv = mvp;
  233|   164k|            mvstack[last].weight = 2;
  234|   164k|            *cnt = last + 1;
  235|   164k|        }
  236|   149k|    }
  237|  2.25M|}
refmvs.c:mv_projection:
  175|  2.93M|static inline union mv mv_projection(const union mv mv, const int num, const int den) {
  176|  2.93M|    static const uint16_t div_mult[32] = {
  177|  2.93M|           0, 16384, 8192, 5461, 4096, 3276, 2730, 2340,
  178|  2.93M|        2048,  1820, 1638, 1489, 1365, 1260, 1170, 1092,
  179|  2.93M|        1024,   963,  910,  862,  819,  780,  744,  712,
  180|  2.93M|         682,   655,  630,  606,  585,  564,  546,  528
  181|  2.93M|    };
  182|  2.93M|    assert(den > 0 && den < 32);
  ------------------
  |  Branch (182:5): [True: 2.93M, False: 18.4E]
  |  Branch (182:5): [True: 2.93M, False: 18.4E]
  ------------------
  183|  2.93M|    assert(num > -32 && num < 32);
  ------------------
  |  Branch (183:5): [True: 2.93M, False: 18.4E]
  |  Branch (183:5): [True: 2.93M, False: 18.4E]
  ------------------
  184|  2.93M|    const int frac = num * div_mult[den];
  185|  2.93M|    const int y = mv.y * frac, x = mv.x * frac;
  186|       |    // Round and clip according to AV1 spec section 7.9.3
  187|  2.93M|    return (union mv) { // 0x3fff == (1 << 14) - 1
  188|  2.93M|        .y = iclip((y + 8192 + (y >> 31)) >> 14, -0x3fff, 0x3fff),
  189|  2.93M|        .x = iclip((x + 8192 + (x >> 31)) >> 14, -0x3fff, 0x3fff)
  190|  2.93M|    };
  191|  2.93M|}
refmvs.c:add_compound_extended_candidate:
  245|   490k|{
  246|   490k|    refmvs_candidate *const diff = &same[2];
  247|   490k|    int *const diff_count = &same_count[2];
  248|       |
  249|  1.24M|    for (int n = 0; n < 2; n++) {
  ------------------
  |  Branch (249:21): [True: 949k, False: 296k]
  ------------------
  250|   949k|        const int cand_ref = cand_b->ref.ref[n];
  251|       |
  252|   949k|        if (cand_ref <= 0) break;
  ------------------
  |  Branch (252:13): [True: 193k, False: 755k]
  ------------------
  253|       |
  254|   755k|        mv cand_mv = cand_b->mv.mv[n];
  255|   755k|        if (cand_ref == ref.ref[0]) {
  ------------------
  |  Branch (255:13): [True: 283k, False: 471k]
  ------------------
  256|   283k|            if (same_count[0] < 2)
  ------------------
  |  Branch (256:17): [True: 272k, False: 11.2k]
  ------------------
  257|   272k|                same[same_count[0]++].mv.mv[0] = cand_mv;
  258|   283k|            if (diff_count[1] < 2) {
  ------------------
  |  Branch (258:17): [True: 234k, False: 49.5k]
  ------------------
  259|   234k|                if (sign1 ^ sign_bias[cand_ref - 1]) {
  ------------------
  |  Branch (259:21): [True: 21.5k, False: 212k]
  ------------------
  260|  21.5k|                    cand_mv.y = -cand_mv.y;
  261|  21.5k|                    cand_mv.x = -cand_mv.x;
  262|  21.5k|                }
  263|   234k|                diff[diff_count[1]++].mv.mv[1] = cand_mv;
  264|   234k|            }
  265|   471k|        } else if (cand_ref == ref.ref[1]) {
  ------------------
  |  Branch (265:20): [True: 254k, False: 216k]
  ------------------
  266|   254k|            if (same_count[1] < 2)
  ------------------
  |  Branch (266:17): [True: 246k, False: 8.28k]
  ------------------
  267|   246k|                same[same_count[1]++].mv.mv[1] = cand_mv;
  268|   254k|            if (diff_count[0] < 2) {
  ------------------
  |  Branch (268:17): [True: 209k, False: 45.4k]
  ------------------
  269|   209k|                if (sign0 ^ sign_bias[cand_ref - 1]) {
  ------------------
  |  Branch (269:21): [True: 22.2k, False: 187k]
  ------------------
  270|  22.2k|                    cand_mv.y = -cand_mv.y;
  271|  22.2k|                    cand_mv.x = -cand_mv.x;
  272|  22.2k|                }
  273|   209k|                diff[diff_count[0]++].mv.mv[0] = cand_mv;
  274|   209k|            }
  275|   254k|        } else {
  276|   216k|            mv i_cand_mv = (union mv) {
  277|   216k|                .x = -cand_mv.x,
  278|   216k|                .y = -cand_mv.y
  279|   216k|            };
  280|       |
  281|   216k|            if (diff_count[0] < 2) {
  ------------------
  |  Branch (281:17): [True: 170k, False: 46.5k]
  ------------------
  282|   170k|                diff[diff_count[0]++].mv.mv[0] =
  283|   170k|                    sign0 ^ sign_bias[cand_ref - 1] ?
  ------------------
  |  Branch (283:21): [True: 2.43k, False: 167k]
  ------------------
  284|   167k|                    i_cand_mv : cand_mv;
  285|   170k|            }
  286|       |
  287|   216k|            if (diff_count[1] < 2) {
  ------------------
  |  Branch (287:17): [True: 154k, False: 61.9k]
  ------------------
  288|   154k|                diff[diff_count[1]++].mv.mv[1] =
  289|   154k|                    sign1 ^ sign_bias[cand_ref - 1] ?
  ------------------
  |  Branch (289:21): [True: 5.06k, False: 149k]
  ------------------
  290|   149k|                    i_cand_mv : cand_mv;
  291|   154k|            }
  292|   216k|        }
  293|   755k|    }
  294|   490k|}
refmvs.c:add_single_extended_candidate:
  299|  3.26M|{
  300|  6.42M|    for (int n = 0; n < 2; n++) {
  ------------------
  |  Branch (300:21): [True: 6.34M, False: 71.5k]
  ------------------
  301|  6.34M|        const int cand_ref = cand_b->ref.ref[n];
  302|       |
  303|  6.34M|        if (cand_ref <= 0) break;
  ------------------
  |  Branch (303:13): [True: 3.19M, False: 3.15M]
  ------------------
  304|       |        // we need to continue even if cand_ref == ref.ref[0], since
  305|       |        // the candidate could have been added as a globalmv variant,
  306|       |        // which changes the value
  307|       |        // FIXME if scan_{row,col}() returned a mask for the nearest
  308|       |        // edge, we could skip the appropriate ones here
  309|       |
  310|  3.15M|        mv cand_mv = cand_b->mv.mv[n];
  311|  3.15M|        if (sign ^ sign_bias[cand_ref - 1]) {
  ------------------
  |  Branch (311:13): [True: 16.9k, False: 3.13M]
  ------------------
  312|  16.9k|            cand_mv.y = -cand_mv.y;
  313|  16.9k|            cand_mv.x = -cand_mv.x;
  314|  16.9k|        }
  315|       |
  316|  3.15M|        int m;
  317|  3.15M|        const int last = *cnt;
  318|  3.45M|        for (m = 0; m < last; m++)
  ------------------
  |  Branch (318:21): [True: 3.09M, False: 366k]
  ------------------
  319|  3.09M|            if (cand_mv.n == mvstack[m].mv.mv[0].n)
  ------------------
  |  Branch (319:17): [True: 2.78M, False: 301k]
  ------------------
  320|  2.78M|                break;
  321|  3.15M|        if (m == last) {
  ------------------
  |  Branch (321:13): [True: 370k, False: 2.78M]
  ------------------
  322|   370k|            mvstack[m].mv.mv[0] = cand_mv;
  323|   370k|            mvstack[m].weight = 2; // "minimal"
  324|   370k|            *cnt = last + 1;
  325|   370k|        }
  326|  3.15M|    }
  327|  3.26M|}

decode.c:dav1d_refmvs_save_tmvs:
  145|   832k|{
  146|   832k|    const refmvs_frame *const rf = rt->rf;
  147|       |
  148|   832k|    assert(row_start8 >= 0);
  ------------------
  |  Branch (148:5): [True: 832k, False: 48]
  ------------------
  149|   832k|    assert((unsigned) (row_end8 - row_start8) <= 16U);
  ------------------
  |  Branch (149:5): [True: 832k, False: 18.4E]
  ------------------
  150|   832k|    row_end8 = imin(row_end8, rf->ih8);
  151|   832k|    col_end8 = imin(col_end8, rf->iw8);
  152|       |
  153|   832k|    const ptrdiff_t stride = rf->rp_stride;
  154|   832k|    const uint8_t *const ref_sign = rf->mfmv_sign;
  155|   832k|    refmvs_temporal_block *rp = &rf->rp[row_start8 * stride];
  156|       |
  157|   832k|    dsp->save_tmvs(rp, stride, rt->r + 6, ref_sign,
  158|   832k|                   col_end8, row_end8, col_start8, row_start8);
  159|   832k|}

dav1d_init_last_nonzero_col_from_eob_tables:
  350|  2.99k|COLD void dav1d_init_last_nonzero_col_from_eob_tables(void) {
  351|       |    static pthread_once_t initted = PTHREAD_ONCE_INIT;
  352|  2.99k|    pthread_once(&initted, init_internal);
  353|  2.99k|}
scan.c:init_internal:
  333|      1|static COLD void init_internal(void) {
  334|      1|    init_tbl(last_nonzero_col_from_eob_4x4,   scan_4x4,    4,  4);
  335|      1|    init_tbl(last_nonzero_col_from_eob_8x8,   scan_8x8,    8,  8);
  336|      1|    init_tbl(last_nonzero_col_from_eob_16x16, scan_16x16, 16, 16);
  337|      1|    init_tbl(last_nonzero_col_from_eob_32x32, scan_32x32, 32, 32);
  338|      1|    init_tbl(last_nonzero_col_from_eob_4x8,   scan_4x8,    4,  8);
  339|      1|    init_tbl(last_nonzero_col_from_eob_8x4,   scan_8x4,    8,  4);
  340|      1|    init_tbl(last_nonzero_col_from_eob_8x16,  scan_8x16,   8, 16);
  341|      1|    init_tbl(last_nonzero_col_from_eob_16x8,  scan_16x8,  16,  8);
  342|      1|    init_tbl(last_nonzero_col_from_eob_16x32, scan_16x32, 16, 32);
  343|      1|    init_tbl(last_nonzero_col_from_eob_32x16, scan_32x16, 32, 16);
  344|      1|    init_tbl(last_nonzero_col_from_eob_4x16,  scan_4x16,   4, 16);
  345|      1|    init_tbl(last_nonzero_col_from_eob_16x4,  scan_16x4,  16,  4);
  346|      1|    init_tbl(last_nonzero_col_from_eob_8x32,  scan_8x32,   8, 32);
  347|      1|    init_tbl(last_nonzero_col_from_eob_32x8,  scan_32x8,  32,  8);
  348|      1|}
scan.c:init_tbl:
  321|     14|{
  322|     14|    int max_col = 0;
  323|    218|    for (int y = 0, n = 0; y < h; y++) {
  ------------------
  |  Branch (323:28): [True: 204, False: 14]
  ------------------
  324|  3.54k|        for (int x = 0; x < w; x++, n++) {
  ------------------
  |  Branch (324:25): [True: 3.34k, False: 204]
  ------------------
  325|  3.34k|            const int rc = scan[n];
  326|  3.34k|            const int rcx = rc & (h - 1);
  327|  3.34k|            max_col = imax(max_col, rcx);
  328|  3.34k|            last_nonzero_col_from_eob[n] = max_col;
  329|  3.34k|        }
  330|    204|    }
  331|     14|}

thread_task.c:dav1d_set_thread_name:
  152|  37.6k|static inline void dav1d_set_thread_name(const char *const name) {
  153|       |    prctl(PR_SET_NAME, name);
  154|  37.6k|}

dav1d_task_create_tile_sbrow:
  270|   589k|{
  271|   589k|    Dav1dTask *tasks = f->task_thread.tile_tasks[0];
  272|   589k|    const int uses_2pass = f->c->n_fc > 1;
  273|   589k|    const int n_tasks_per_pass = f->frame_hdr->tiling.cols * f->frame_hdr->tiling.rows;
  274|   589k|    const int n_tasks = n_tasks_per_pass * (1 + uses_2pass);
  275|   589k|    if (pass < 2) {
  ------------------
  |  Branch (275:9): [True: 294k, False: 294k]
  ------------------
  276|   294k|        if (n_tasks > f->task_thread.num_tile_tasks) {
  ------------------
  |  Branch (276:13): [True: 16.0k, False: 278k]
  ------------------
  277|  16.0k|            const size_t size = sizeof(Dav1dTask) * n_tasks;
  278|  16.0k|            tasks = dav1d_realloc(ALLOC_COMMON_CTX, f->task_thread.tile_tasks[0], size);
  ------------------
  |  |  133|  16.0k|#define dav1d_realloc(type, ptr, sz) realloc(ptr, sz)
  ------------------
  279|  16.0k|            if (!tasks) return -1;
  ------------------
  |  Branch (279:17): [True: 0, False: 16.0k]
  ------------------
  280|  16.0k|            memset(tasks, 0, size);
  281|  16.0k|            f->task_thread.tile_tasks[0] = tasks;
  282|  16.0k|            f->task_thread.num_tile_tasks = n_tasks;
  283|  16.0k|        }
  284|   294k|        f->task_thread.tile_tasks[1] = tasks + n_tasks_per_pass;
  285|   294k|    }
  286|   589k|    assert(n_tasks <= f->task_thread.num_tile_tasks);
  ------------------
  |  Branch (286:5): [True: 589k, False: 66]
  ------------------
  287|       |
  288|   589k|    Dav1dTask *pf_t;
  289|   589k|    if (create_filter_sbrow(f, pass, &pf_t))
  ------------------
  |  Branch (289:9): [True: 0, False: 589k]
  ------------------
  290|      0|        return -1;
  291|       |
  292|   589k|    Dav1dTask *const p1_tasks = f->task_thread.tile_tasks[1];
  293|   589k|    Dav1dTask *prev_t = NULL;
  294|   589k|    if (pass == 2) {
  ------------------
  |  Branch (294:9): [True: 295k, False: 294k]
  ------------------
  295|   295k|        prev_t = &p1_tasks[n_tasks_per_pass - 1];
  296|       |        // PF task is scheduled after the last sby=0 TILE task
  297|   295k|        if (f->frame_hdr->tiling.rows == 1)
  ------------------
  |  Branch (297:13): [True: 287k, False: 7.35k]
  ------------------
  298|   287k|            prev_t = prev_t->next;
  299|   295k|    }
  300|   589k|    tasks += (pass & 1) * n_tasks_per_pass;
  301|  1.21M|    for (int tile_idx = 0; tile_idx < n_tasks_per_pass; tile_idx++) {
  ------------------
  |  Branch (301:28): [True: 629k, False: 589k]
  ------------------
  302|   629k|        Dav1dTileState *const ts = &f->ts[tile_idx];
  303|   629k|        Dav1dTask *t = &tasks[tile_idx];
  304|   629k|        t->sby = ts->tiling.row_start >> f->sb_shift;
  305|   629k|        if (pf_t && t->sby) {
  ------------------
  |  Branch (305:13): [True: 616k, False: 13.0k]
  |  Branch (305:21): [True: 14.7k, False: 602k]
  ------------------
  306|  14.7k|            prev_t->next = pf_t;
  307|  14.7k|            prev_t = pf_t;
  308|  14.7k|            pf_t = NULL;
  309|  14.7k|        }
  310|   629k|        t->recon_progress = 0;
  311|   629k|        t->deblock_progress = 0;
  312|   629k|        t->deps_skip = 0;
  313|   629k|        t->type = pass != 1 ? DAV1D_TASK_TYPE_TILE_RECONSTRUCTION :
  ------------------
  |  Branch (313:19): [True: 314k, False: 314k]
  ------------------
  314|   629k|                              DAV1D_TASK_TYPE_TILE_ENTROPY;
  315|   629k|        t->frame_idx = (int)(f - f->c->fc);
  316|   629k|        if (prev_t) prev_t->next = t;
  ------------------
  |  Branch (316:13): [True: 335k, False: 294k]
  ------------------
  317|   629k|        prev_t = t;
  318|   629k|    }
  319|   589k|    if (pf_t) {
  ------------------
  |  Branch (319:9): [True: 575k, False: 14.1k]
  ------------------
  320|   575k|        prev_t->next = pf_t;
  321|   575k|        prev_t = pf_t;
  322|   575k|    }
  323|   589k|    prev_t->next = NULL;
  324|       |
  325|   589k|    atomic_store(&f->task_thread.done[pass & 1], 0);
  326|       |
  327|       |    // XXX in theory this could be done locklessly, at this point they are no
  328|       |    // tasks in the frameQ, so no other runner should be using this lock, but
  329|       |    // we must add both passes at once
  330|   589k|    if (!(pass & 1)) {
  ------------------
  |  Branch (330:9): [True: 295k, False: 294k]
  ------------------
  331|   295k|        pthread_mutex_lock(&f->task_thread.pending_tasks.lock);
  332|   295k|        assert(f->task_thread.pending_tasks.head == NULL);
  ------------------
  |  Branch (332:9): [True: 295k, False: 18.4E]
  ------------------
  333|   295k|        f->task_thread.pending_tasks.head = f->task_thread.tile_tasks[pass == 2];
  334|   295k|        f->task_thread.pending_tasks.tail = prev_t;
  335|   295k|        atomic_store(&f->task_thread.pending_tasks.merge, 1);
  336|   295k|        atomic_store(&f->task_thread.init_done, 1);
  337|   295k|        pthread_mutex_unlock(&f->task_thread.pending_tasks.lock);
  338|   295k|    }
  339|   589k|    return 0;
  340|   589k|}
dav1d_task_frame_init:
  342|   356k|void dav1d_task_frame_init(Dav1dFrameContext *const f) {
  343|   356k|    const Dav1dContext *const c = f->c;
  344|       |
  345|   356k|    atomic_store(&f->task_thread.init_done, 0);
  346|       |    // schedule init task, which will schedule the remaining tasks
  347|   356k|    Dav1dTask *const t = &f->task_thread.init_task;
  348|   356k|    t->type = DAV1D_TASK_TYPE_INIT;
  349|   356k|    t->frame_idx = (int)(f - c->fc);
  350|   356k|    t->sby = 0;
  351|   356k|    t->recon_progress = t->deblock_progress = 0;
  352|   356k|    insert_task(f, t, 1);
  353|   356k|}
dav1d_task_delayed_fg:
  357|  10.1k|{
  358|  10.1k|    struct TaskThreadData *const ttd = &c->task_thread;
  359|  10.1k|    ttd->delayed_fg.in = in;
  360|  10.1k|    ttd->delayed_fg.out = out;
  361|  10.1k|    ttd->delayed_fg.type = DAV1D_TASK_TYPE_FG_PREP;
  362|  10.1k|    atomic_init(&ttd->delayed_fg.progress[0], 0);
  363|  10.1k|    atomic_init(&ttd->delayed_fg.progress[1], 0);
  364|  10.1k|    pthread_mutex_lock(&ttd->lock);
  365|  10.1k|    ttd->delayed_fg.exec = 1;
  366|  10.1k|    ttd->delayed_fg.finished = 0;
  367|  10.1k|    pthread_cond_signal(&ttd->cond);
  368|  10.1k|    do {
  369|  10.1k|        pthread_cond_wait(&ttd->delayed_fg.cond, &ttd->lock);
  370|  10.1k|    } while (!ttd->delayed_fg.finished);
  ------------------
  |  Branch (370:14): [True: 0, False: 10.1k]
  ------------------
  371|  10.1k|    pthread_mutex_unlock(&ttd->lock);
  372|  10.1k|}
dav1d_worker_task:
  556|  37.6k|void *dav1d_worker_task(void *data) {
  557|  37.6k|    Dav1dTaskContext *const tc = data;
  558|  37.6k|    const Dav1dContext *const c = tc->c;
  559|  37.6k|    struct TaskThreadData *const ttd = tc->task_thread.ttd;
  560|       |
  561|  37.6k|    dav1d_set_thread_name("dav1d-worker");
  562|       |
  563|  37.6k|    pthread_mutex_lock(&ttd->lock);
  564|  21.7M|    for (;;) {
  565|  21.7M|        if (tc->task_thread.die) break;
  ------------------
  |  Branch (565:13): [True: 37.6k, False: 21.6M]
  ------------------
  566|  21.6M|        if (atomic_load(c->flush)) goto park;
  ------------------
  |  Branch (566:13): [True: 9.51k, False: 21.6M]
  ------------------
  567|       |
  568|  21.6M|        merge_pending(c);
  569|  21.6M|        if (ttd->delayed_fg.exec) { // run delayed film grain first
  ------------------
  |  Branch (569:13): [True: 16.2k, False: 21.6M]
  ------------------
  570|  16.2k|            delayed_fg_task(c, ttd);
  571|  16.2k|            continue;
  572|  16.2k|        }
  573|  21.6M|        Dav1dFrameContext *f;
  574|  21.6M|        Dav1dTask *t, *prev_t = NULL;
  575|  21.6M|        if (c->n_fc > 1) { // run init tasks second
  ------------------
  |  Branch (575:13): [True: 21.6M, False: 0]
  ------------------
  576|   107M|            for (unsigned i = 0; i < c->n_fc; i++) {
  ------------------
  |  Branch (576:34): [True: 86.2M, False: 21.2M]
  ------------------
  577|  86.2M|                const unsigned first = atomic_load(&ttd->first);
  578|  86.2M|                f = &c->fc[(first + i) % c->n_fc];
  579|  86.2M|                if (atomic_load(&f->task_thread.init_done)) continue;
  ------------------
  |  Branch (579:21): [True: 77.0M, False: 9.29M]
  ------------------
  580|  9.29M|                t = f->task_thread.task_head;
  581|  9.29M|                if (!t) continue;
  ------------------
  |  Branch (581:21): [True: 7.80M, False: 1.48M]
  ------------------
  582|  1.48M|                if (t->type == DAV1D_TASK_TYPE_INIT) goto found;
  ------------------
  |  Branch (582:21): [True: 355k, False: 1.13M]
  ------------------
  583|  1.13M|                if (t->type == DAV1D_TASK_TYPE_INIT_CDF) {
  ------------------
  |  Branch (583:21): [True: 1.13M, False: 0]
  ------------------
  584|       |                    // XXX This can be a simple else, if adding tasks of both
  585|       |                    // passes at once (in dav1d_task_create_tile_sbrow).
  586|       |                    // Adding the tasks to the pending Q can result in a
  587|       |                    // thread merging them before setting init_done.
  588|       |                    // We will need to set init_done before adding to the
  589|       |                    // pending Q, so maybe return the tasks, set init_done,
  590|       |                    // and add to pending Q only then.
  591|  1.13M|                    const int p1 = f->in_cdf.progress ?
  ------------------
  |  Branch (591:36): [True: 1.13M, False: 0]
  ------------------
  592|  1.13M|                        atomic_load(f->in_cdf.progress) : 1;
  593|  1.13M|                    if (p1) {
  ------------------
  |  Branch (593:25): [True: 11.0k, False: 1.12M]
  ------------------
  594|  11.0k|                        atomic_fetch_or(&f->task_thread.error, p1 == TILE_ERROR);
  595|  11.0k|                        goto found;
  596|  11.0k|                    }
  597|  1.13M|                }
  598|  1.13M|            }
  599|  21.6M|        }
  600|  24.0M|        while (ttd->cur < c->n_fc) { // run decoding tasks last
  ------------------
  |  Branch (600:16): [True: 23.1M, False: 918k]
  ------------------
  601|  23.1M|            const unsigned first = atomic_load(&ttd->first);
  602|  23.1M|            f = &c->fc[(first + ttd->cur) % c->n_fc];
  603|  23.1M|            merge_pending_frame(f);
  604|  23.1M|            prev_t = f->task_thread.task_cur_prev;
  605|  23.1M|            t = prev_t ? prev_t->next : f->task_thread.task_head;
  ------------------
  |  Branch (605:17): [True: 150k, False: 23.0M]
  ------------------
  606|  28.9M|            while (t) {
  ------------------
  |  Branch (606:20): [True: 26.1M, False: 2.80M]
  ------------------
  607|  26.1M|                if (t->type == DAV1D_TASK_TYPE_INIT_CDF) goto next;
  ------------------
  |  Branch (607:21): [True: 27.5k, False: 26.0M]
  ------------------
  608|  26.0M|                else if (t->type == DAV1D_TASK_TYPE_TILE_ENTROPY ||
  ------------------
  |  Branch (608:26): [True: 2.27M, False: 23.8M]
  ------------------
  609|  23.8M|                         t->type == DAV1D_TASK_TYPE_TILE_RECONSTRUCTION)
  ------------------
  |  Branch (609:26): [True: 2.55M, False: 21.2M]
  ------------------
  610|  4.82M|                {
  611|       |                    // if not bottom sbrow of tile, this task will be re-added
  612|       |                    // after it's finished
  613|  4.82M|                    if (!check_tile(t, f, c->n_fc > 1))
  ------------------
  |  Branch (613:25): [True: 4.46M, False: 359k]
  ------------------
  614|  4.46M|                        goto found;
  615|  21.2M|                } else if (t->recon_progress) {
  ------------------
  |  Branch (615:28): [True: 19.5M, False: 1.73M]
  ------------------
  616|  19.5M|                    const int p = t->type == DAV1D_TASK_TYPE_ENTROPY_PROGRESS;
  617|  19.5M|                    int error = atomic_load(&f->task_thread.error);
  618|  19.5M|                    assert(!atomic_load(&f->task_thread.done[p]) || error);
  ------------------
  |  Branch (618:21): [True: 13.5M, False: 5.97M]
  |  Branch (618:21): [True: 5.97M, False: 0]
  ------------------
  619|  19.5M|                    const int tile_row_base = f->frame_hdr->tiling.cols *
  620|  19.5M|                                              f->frame_thread.next_tile_row[p];
  621|  19.5M|                    if (p) {
  ------------------
  |  Branch (621:25): [True: 8.28M, False: 11.2M]
  ------------------
  622|  8.28M|                        atomic_int *const prog = &f->frame_thread.entropy_progress;
  623|  8.28M|                        const int p1 = atomic_load(prog);
  624|  8.28M|                        if (p1 < t->sby) goto next;
  ------------------
  |  Branch (624:29): [True: 14.6k, False: 8.26M]
  ------------------
  625|  8.28M|                        atomic_fetch_or(&f->task_thread.error, p1 == TILE_ERROR);
  626|  8.26M|                    }
  627|  36.4M|                    for (int tc = 0; tc < f->frame_hdr->tiling.cols; tc++) {
  ------------------
  |  Branch (627:38): [True: 20.7M, False: 15.7M]
  ------------------
  628|  20.7M|                        Dav1dTileState *const ts = &f->ts[tile_row_base + tc];
  629|  20.7M|                        const int p2 = atomic_load(&ts->progress[p]);
  630|  20.7M|                        if (p2 < t->recon_progress) goto next;
  ------------------
  |  Branch (630:29): [True: 3.79M, False: 16.9M]
  ------------------
  631|  20.7M|                        atomic_fetch_or(&f->task_thread.error, p2 == TILE_ERROR);
  632|  16.9M|                    }
  633|  15.7M|                    if (t->sby + 1 < f->sbh) {
  ------------------
  |  Branch (633:25): [True: 15.1M, False: 588k]
  ------------------
  634|       |                        // add sby+1 to list to replace this one
  635|  15.1M|                        Dav1dTask *next_t = &t[1];
  636|  15.1M|                        *next_t = *t;
  637|  15.1M|                        next_t->sby++;
  638|  15.1M|                        const int ntr = f->frame_thread.next_tile_row[p] + 1;
  639|  15.1M|                        const int start = f->frame_hdr->tiling.row_start_sb[ntr];
  640|  15.1M|                        if (next_t->sby == start)
  ------------------
  |  Branch (640:29): [True: 22.6k, False: 15.1M]
  ------------------
  641|  22.6k|                            f->frame_thread.next_tile_row[p] = ntr;
  642|  15.1M|                        next_t->recon_progress = next_t->sby + 1;
  643|  15.1M|                        insert_task(f, next_t, 0);
  644|  15.1M|                    }
  645|  15.7M|                    goto found;
  646|  19.5M|                } else if (t->type == DAV1D_TASK_TYPE_CDEF) {
  ------------------
  |  Branch (646:28): [True: 1.40M, False: 323k]
  ------------------
  647|  1.40M|                    atomic_uint *prog = f->frame_thread.copy_lpf_progress;
  648|  1.40M|                    const int p1 = atomic_load(&prog[(t->sby - 1) >> 5]);
  649|  1.40M|                    if (p1 & (1U << ((t->sby - 1) & 31)))
  ------------------
  |  Branch (649:25): [True: 150k, False: 1.25M]
  ------------------
  650|   150k|                        goto found;
  651|  1.40M|                } else {
  652|   323k|                    assert(t->deblock_progress);
  ------------------
  |  Branch (652:21): [True: 323k, False: 0]
  ------------------
  653|   323k|                    const int p1 = atomic_load(&f->frame_thread.deblock_progress);
  654|   323k|                    if (p1 >= t->deblock_progress) {
  ------------------
  |  Branch (654:25): [True: 18.4k, False: 304k]
  ------------------
  655|  18.4k|                        atomic_fetch_or(&f->task_thread.error, p1 == TILE_ERROR);
  656|  18.4k|                        goto found;
  657|  18.4k|                    }
  658|   323k|                }
  659|  5.75M|            next:
  660|  5.75M|                prev_t = t;
  661|  5.75M|                t = t->next;
  662|  5.75M|                f->task_thread.task_cur_prev = prev_t;
  663|  5.75M|            }
  664|  2.80M|            ttd->cur++;
  665|  2.80M|        }
  666|   918k|        if (reset_task_cur(c, ttd, UINT_MAX)) continue;
  ------------------
  |  Branch (666:13): [True: 9.94k, False: 908k]
  ------------------
  667|   908k|        if (merge_pending(c)) continue;
  ------------------
  |  Branch (667:13): [True: 5.02k, False: 903k]
  ------------------
  668|   913k|    park:
  669|   913k|        tc->task_thread.flushed = 1;
  670|   913k|        pthread_cond_signal(&tc->task_thread.td.cond);
  671|       |        // we want to be woken up next time progress is signaled
  672|   913k|        atomic_store(&ttd->cond_signaled, 0);
  673|   913k|        pthread_cond_wait(&ttd->cond, &ttd->lock);
  674|   913k|        tc->task_thread.flushed = 0;
  675|   913k|        reset_task_cur(c, ttd, UINT_MAX);
  676|   913k|        continue;
  677|       |
  678|  20.7M|    found:
  679|       |        // remove t from list
  680|  20.7M|        if (prev_t) prev_t->next = t->next;
  ------------------
  |  Branch (680:13): [True: 3.08M, False: 17.6M]
  ------------------
  681|  17.6M|        else f->task_thread.task_head = t->next;
  682|  20.7M|        if (!t->next) f->task_thread.task_tail = prev_t;
  ------------------
  |  Branch (682:13): [True: 938k, False: 19.7M]
  ------------------
  683|  20.7M|        if (t->type > DAV1D_TASK_TYPE_INIT_CDF && !f->task_thread.task_head)
  ------------------
  |  Branch (683:13): [True: 20.3M, False: 366k]
  |  Branch (683:51): [True: 294k, False: 20.0M]
  ------------------
  684|   294k|            ttd->cur++;
  685|  20.7M|        t->next = NULL;
  686|       |        // we don't need to check cond_signaled here, since we found a task
  687|       |        // after the last signal so we want to re-signal the next waiting thread
  688|       |        // and again won't need to signal after that
  689|  20.7M|        atomic_store(&ttd->cond_signaled, 1);
  690|  20.7M|        pthread_cond_signal(&ttd->cond);
  691|  20.7M|        pthread_mutex_unlock(&ttd->lock);
  692|  33.1M|    found_unlocked:;
  693|  33.1M|        const int flush = atomic_load(c->flush);
  694|  33.1M|        int error = atomic_fetch_or(&f->task_thread.error, flush) | flush;
  695|       |
  696|       |        // run it
  697|  33.1M|        tc->f = f;
  698|  33.1M|        int sby = t->sby;
  699|  33.1M|        switch (t->type) {
  700|   355k|        case DAV1D_TASK_TYPE_INIT: {
  ------------------
  |  Branch (700:9): [True: 355k, False: 32.7M]
  ------------------
  701|   355k|            assert(c->n_fc > 1);
  ------------------
  |  Branch (701:13): [True: 355k, False: 8]
  ------------------
  702|   355k|            int res = dav1d_decode_frame_init(f);
  703|   355k|            int p1 = f->in_cdf.progress ? atomic_load(f->in_cdf.progress) : 1;
  ------------------
  |  Branch (703:22): [True: 69.4k, False: 285k]
  ------------------
  704|   355k|            if (res || p1 == TILE_ERROR) {
  ------------------
  |  |   36|   354k|#define TILE_ERROR (INT_MAX - 1)
  ------------------
  |  Branch (704:17): [True: 653, False: 354k]
  |  Branch (704:24): [True: 46.4k, False: 308k]
  ------------------
  705|  46.5k|                pthread_mutex_lock(&ttd->lock);
  706|  46.5k|                abort_frame(f, res ? res : DAV1D_ERR(EINVAL));
  ------------------
  |  |   58|  46.5k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (706:32): [True: 0, False: 46.5k]
  ------------------
  707|  46.5k|                reset_task_cur(c, ttd, t->frame_idx);
  708|   308k|            } else {
  709|   308k|                t->type = DAV1D_TASK_TYPE_INIT_CDF;
  710|   308k|                if (p1) goto found_unlocked;
  ------------------
  |  Branch (710:21): [True: 297k, False: 11.7k]
  ------------------
  711|  11.7k|                add_pending(f, t);
  712|  11.7k|                pthread_mutex_lock(&ttd->lock);
  713|  11.7k|            }
  714|  58.2k|            continue;
  715|   355k|        }
  716|   308k|        case DAV1D_TASK_TYPE_INIT_CDF: {
  ------------------
  |  Branch (716:9): [True: 308k, False: 32.8M]
  ------------------
  717|   308k|            assert(c->n_fc > 1);
  ------------------
  |  Branch (717:13): [True: 308k, False: 18.4E]
  ------------------
  718|   308k|            int res = DAV1D_ERR(EINVAL);
  ------------------
  |  |   58|   308k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  719|   308k|            if (!atomic_load(&f->task_thread.error))
  ------------------
  |  Branch (719:17): [True: 302k, False: 6.07k]
  ------------------
  720|   302k|                res = dav1d_decode_frame_init_cdf(f);
  721|   308k|            if (f->frame_hdr->refresh_context && !f->task_thread.update_set)
  ------------------
  |  Branch (721:17): [True: 26.7k, False: 282k]
  |  Branch (721:50): [True: 2.14k, False: 24.5k]
  ------------------
  722|   308k|                atomic_store(f->out_cdf.progress, res < 0 ? TILE_ERROR : 1);
  ------------------
  |  Branch (722:17): [True: 2.14k, False: 0]
  ------------------
  723|   898k|            for (int p = 1; p <= 2 && !res; p++)
  ------------------
  |  Branch (723:29): [True: 602k, False: 295k]
  |  Branch (723:39): [True: 589k, False: 13.5k]
  ------------------
  724|   589k|                res = dav1d_task_create_tile_sbrow(f, p, 0);
  725|   308k|            pthread_mutex_lock(&ttd->lock);
  726|   308k|            if (res) {
  ------------------
  |  Branch (726:17): [True: 13.5k, False: 295k]
  ------------------
  727|  13.5k|                abort_frame(f, DAV1D_ERR(ENOMEM));
  ------------------
  |  |   58|  13.5k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  728|  13.5k|                reset_task_cur(c, ttd, t->frame_idx);
  729|  13.5k|                atomic_store(&f->task_thread.init_done, 1);
  730|  13.5k|            }
  731|   308k|            continue;
  732|   308k|        }
  733|  8.29M|        case DAV1D_TASK_TYPE_TILE_ENTROPY:
  ------------------
  |  Branch (733:9): [True: 8.29M, False: 24.8M]
  ------------------
  734|  16.5M|        case DAV1D_TASK_TYPE_TILE_RECONSTRUCTION: {
  ------------------
  |  Branch (734:9): [True: 8.28M, False: 24.8M]
  ------------------
  735|  16.5M|            const int p = t->type == DAV1D_TASK_TYPE_TILE_ENTROPY;
  736|  16.5M|            const int tile_idx = (int)(t - f->task_thread.tile_tasks[p]);
  737|  16.5M|            Dav1dTileState *const ts = &f->ts[tile_idx];
  738|       |
  739|  16.5M|            tc->ts = ts;
  740|  16.5M|            tc->by = sby << f->sb_shift;
  741|  16.5M|            const int uses_2pass = c->n_fc > 1;
  742|  16.5M|            tc->frame_thread.pass = !uses_2pass ? 0 :
  ------------------
  |  Branch (742:37): [True: 0, False: 16.5M]
  ------------------
  743|  16.5M|                1 + (t->type == DAV1D_TASK_TYPE_TILE_RECONSTRUCTION);
  744|  16.5M|            if (!error) error = dav1d_decode_tile_sbrow(tc);
  ------------------
  |  Branch (744:17): [True: 4.12M, False: 12.4M]
  ------------------
  745|  16.5M|            const int progress = error ? TILE_ERROR : 1 + sby;
  ------------------
  |  |   36|  12.5M|#define TILE_ERROR (INT_MAX - 1)
  ------------------
  |  Branch (745:34): [True: 12.5M, False: 4.03M]
  ------------------
  746|       |
  747|       |            // signal progress
  748|  16.5M|            atomic_fetch_or(&f->task_thread.error, error);
  749|  16.5M|            if (((sby + 1) << f->sb_shift) < ts->tiling.row_end) {
  ------------------
  |  Branch (749:17): [True: 15.9M, False: 642k]
  ------------------
  750|  15.9M|                t->sby++;
  751|  15.9M|                t->deps_skip = 0;
  752|  15.9M|                if (!check_tile(t, f, uses_2pass)) {
  ------------------
  |  Branch (752:21): [True: 12.0M, False: 3.84M]
  ------------------
  753|  12.0M|                    atomic_store(&ts->progress[p], progress);
  754|  12.0M|                    reset_task_cur_async(ttd, t->frame_idx, c->n_fc);
  755|  12.0M|                    if (!atomic_fetch_or(&ttd->cond_signaled, 1))
  ------------------
  |  Branch (755:25): [True: 104, False: 12.0M]
  ------------------
  756|    104|                        pthread_cond_signal(&ttd->cond);
  757|  12.0M|                    goto found_unlocked;
  758|  12.0M|                }
  759|  15.9M|                atomic_store(&ts->progress[p], progress);
  760|  3.84M|                add_pending(f, t);
  761|  3.84M|                pthread_mutex_lock(&ttd->lock);
  762|  3.84M|            } else {
  763|   642k|                pthread_mutex_lock(&ttd->lock);
  764|   642k|                atomic_store(&ts->progress[p], progress);
  765|   642k|                reset_task_cur(c, ttd, t->frame_idx);
  766|   642k|                error = atomic_load(&f->task_thread.error);
  767|   642k|                if (f->frame_hdr->refresh_context &&
  ------------------
  |  Branch (767:21): [True: 53.0k, False: 589k]
  ------------------
  768|  53.0k|                    tc->frame_thread.pass <= 1 && f->task_thread.update_set &&
  ------------------
  |  Branch (768:21): [True: 26.6k, False: 26.4k]
  |  Branch (768:51): [True: 26.6k, False: 0]
  ------------------
  769|  26.6k|                    f->frame_hdr->tiling.update == tile_idx)
  ------------------
  |  Branch (769:21): [True: 20.6k, False: 5.91k]
  ------------------
  770|  20.6k|                {
  771|  20.6k|                    if (!error)
  ------------------
  |  Branch (771:25): [True: 9.85k, False: 10.8k]
  ------------------
  772|  9.85k|                        dav1d_cdf_thread_update(f->frame_hdr, f->out_cdf.data.cdf,
  773|  9.85k|                                                &f->ts[f->frame_hdr->tiling.update].cdf);
  774|  20.6k|                    if (c->n_fc > 1)
  ------------------
  |  Branch (774:25): [True: 20.6k, False: 0]
  ------------------
  775|  20.6k|                        atomic_store(f->out_cdf.progress, error ? TILE_ERROR : 1);
  ------------------
  |  Branch (775:25): [True: 10.8k, False: 9.85k]
  ------------------
  776|  20.6k|                }
  777|   642k|                if (atomic_fetch_sub(&f->task_thread.task_counter, 1) - 1 == 0 &&
  ------------------
  |  Branch (777:21): [True: 8.58k, False: 633k]
  ------------------
  778|   642k|                    atomic_load(&f->task_thread.done[0]) &&
  ------------------
  |  Branch (778:21): [True: 8.58k, False: 0]
  ------------------
  779|  8.58k|                    (!uses_2pass || atomic_load(&f->task_thread.done[1])))
  ------------------
  |  Branch (779:22): [True: 0, False: 8.58k]
  |  Branch (779:37): [True: 8.58k, False: 0]
  ------------------
  780|  8.58k|                {
  781|  8.58k|                    error = atomic_load(&f->task_thread.error);
  782|  8.58k|                    dav1d_decode_frame_exit(f, error == 1 ? DAV1D_ERR(EINVAL) :
  ------------------
  |  |   58|  8.58k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (782:48): [True: 8.58k, False: 0]
  ------------------
  783|  8.58k|                                            error ? DAV1D_ERR(ENOMEM) : 0);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (783:45): [True: 0, False: 0]
  ------------------
  784|  8.58k|                    f->n_tile_data = 0;
  785|  8.58k|                    pthread_cond_signal(&f->task_thread.cond);
  786|  8.58k|                }
  787|   642k|                assert(atomic_load(&f->task_thread.task_counter) >= 0);
  ------------------
  |  Branch (787:17): [True: 628k, False: 14.1k]
  ------------------
  788|   628k|                if (!atomic_fetch_or(&ttd->cond_signaled, 1))
  ------------------
  |  Branch (788:21): [True: 108k, False: 519k]
  ------------------
  789|   108k|                    pthread_cond_signal(&ttd->cond);
  790|   628k|            }
  791|  4.47M|            continue;
  792|  16.5M|        }
  793|  4.47M|        case DAV1D_TASK_TYPE_DEBLOCK_COLS:
  ------------------
  |  Branch (793:9): [True: 2.03M, False: 31.0M]
  ------------------
  794|  2.03M|            if (!atomic_load(&f->task_thread.error))
  ------------------
  |  Branch (794:17): [True: 1.22M, False: 810k]
  ------------------
  795|  1.22M|                f->bd_fn.filter_sbrow_deblock_cols(f, sby);
  796|  2.03M|            if (ensure_progress(ttd, f, t, DAV1D_TASK_TYPE_DEBLOCK_ROWS,
  ------------------
  |  Branch (796:17): [True: 18.4k, False: 2.01M]
  ------------------
  797|  2.03M|                                &f->frame_thread.deblock_progress,
  798|  2.03M|                                &t->deblock_progress)) continue;
  799|       |            // fall-through
  800|  7.40M|        case DAV1D_TASK_TYPE_DEBLOCK_ROWS:
  ------------------
  |  Branch (800:9): [True: 5.39M, False: 27.7M]
  ------------------
  801|  7.40M|            if (!atomic_load(&f->task_thread.error))
  ------------------
  |  Branch (801:17): [True: 1.49M, False: 5.90M]
  ------------------
  802|  1.49M|                f->bd_fn.filter_sbrow_deblock_rows(f, sby);
  803|       |            // signal deblock progress
  804|  7.40M|            if (f->frame_hdr->loopfilter.level_y[0] ||
  ------------------
  |  Branch (804:17): [True: 1.72M, False: 5.68M]
  ------------------
  805|  5.68M|                f->frame_hdr->loopfilter.level_y[1])
  ------------------
  |  Branch (805:17): [True: 306k, False: 5.37M]
  ------------------
  806|  2.03M|            {
  807|  2.03M|                error = atomic_load(&f->task_thread.error);
  808|  2.03M|                atomic_store(&f->frame_thread.deblock_progress,
  ------------------
  |  Branch (808:17): [True: 810k, False: 1.22M]
  ------------------
  809|  2.03M|                             error ? TILE_ERROR : sby + 1);
  810|  2.03M|                reset_task_cur_async(ttd, t->frame_idx, c->n_fc);
  811|  2.03M|                if (!atomic_fetch_or(&ttd->cond_signaled, 1))
  ------------------
  |  Branch (811:21): [True: 10.8k, False: 2.01M]
  ------------------
  812|  10.8k|                    pthread_cond_signal(&ttd->cond);
  813|  5.37M|            } else if (f->seq_hdr->cdef || f->lf.restore_planes) {
  ------------------
  |  Branch (813:24): [True: 5.31M, False: 66.2k]
  |  Branch (813:44): [True: 66.2k, False: 0]
  ------------------
  814|  5.37M|                atomic_fetch_or(&f->frame_thread.copy_lpf_progress[sby >> 5],
  815|  5.37M|                                1U << (sby & 31));
  816|       |                // CDEF needs the top buffer to be saved by lr_copy_lpf of the
  817|       |                // previous sbrow
  818|  5.37M|                if (sby) {
  ------------------
  |  Branch (818:21): [True: 5.19M, False: 178k]
  ------------------
  819|  5.19M|                    int prog = atomic_load(&f->frame_thread.copy_lpf_progress[(sby - 1) >> 5]);
  820|  5.19M|                    if (~prog & (1U << ((sby - 1) & 31))) {
  ------------------
  |  Branch (820:25): [True: 150k, False: 5.04M]
  ------------------
  821|   150k|                        t->type = DAV1D_TASK_TYPE_CDEF;
  822|   150k|                        t->recon_progress = t->deblock_progress = 0;
  823|   150k|                        add_pending(f, t);
  824|   150k|                        pthread_mutex_lock(&ttd->lock);
  825|   150k|                        continue;
  826|   150k|                    }
  827|  5.19M|                }
  828|  5.37M|            }
  829|       |            // fall-through
  830|  7.40M|        case DAV1D_TASK_TYPE_CDEF:
  ------------------
  |  Branch (830:9): [True: 150k, False: 32.9M]
  ------------------
  831|  7.40M|            if (f->seq_hdr->cdef) {
  ------------------
  |  Branch (831:17): [True: 6.11M, False: 1.29M]
  ------------------
  832|  6.11M|                if (!atomic_load(&f->task_thread.error))
  ------------------
  |  Branch (832:21): [True: 932k, False: 5.18M]
  ------------------
  833|   932k|                    f->bd_fn.filter_sbrow_cdef(tc, sby);
  834|  6.11M|                reset_task_cur_async(ttd, t->frame_idx, c->n_fc);
  835|  6.11M|                if (!atomic_fetch_or(&ttd->cond_signaled, 1))
  ------------------
  |  Branch (835:21): [True: 31.4k, False: 6.08M]
  ------------------
  836|  31.4k|                    pthread_cond_signal(&ttd->cond);
  837|  6.11M|            }
  838|       |            // fall-through
  839|  7.41M|        case DAV1D_TASK_TYPE_SUPER_RESOLUTION:
  ------------------
  |  Branch (839:9): [True: 3.25k, False: 33.1M]
  ------------------
  840|  7.41M|            if (f->frame_hdr->width[0] != f->frame_hdr->width[1])
  ------------------
  |  Branch (840:17): [True: 587k, False: 6.82M]
  ------------------
  841|   587k|                if (!atomic_load(&f->task_thread.error))
  ------------------
  |  Branch (841:21): [True: 43.9k, False: 543k]
  ------------------
  842|  43.9k|                    f->bd_fn.filter_sbrow_resize(f, sby);
  843|       |            // fall-through
  844|  7.41M|        case DAV1D_TASK_TYPE_LOOP_RESTORATION:
  ------------------
  |  Branch (844:9): [True: 0, False: 33.1M]
  ------------------
  845|  7.41M|            if (!atomic_load(&f->task_thread.error) && f->lf.restore_planes)
  ------------------
  |  Branch (845:17): [True: 1.49M, False: 5.91M]
  |  Branch (845:56): [True: 217k, False: 1.28M]
  ------------------
  846|   217k|                f->bd_fn.filter_sbrow_lr(f, sby);
  847|       |            // fall-through
  848|  7.86M|        case DAV1D_TASK_TYPE_RECONSTRUCTION_PROGRESS:
  ------------------
  |  Branch (848:9): [True: 450k, False: 32.6M]
  ------------------
  849|       |            // dummy to cover for no post-filters
  850|  15.7M|        case DAV1D_TASK_TYPE_ENTROPY_PROGRESS:
  ------------------
  |  Branch (850:9): [True: 7.86M, False: 25.2M]
  ------------------
  851|       |            // dummy to convert tile progress to frame
  852|  15.7M|            break;
  853|      0|        default: abort();
  ------------------
  |  Branch (853:9): [True: 0, False: 33.1M]
  ------------------
  854|  33.1M|        }
  855|       |        // if task completed [typically LR], signal picture progress as per below
  856|  15.7M|        const int uses_2pass = c->n_fc > 1;
  857|  15.7M|        const int sbh = f->sbh;
  858|  15.7M|        const int sbsz = f->sb_step * 4;
  859|  15.7M|        if (t->type == DAV1D_TASK_TYPE_ENTROPY_PROGRESS) {
  ------------------
  |  Branch (859:13): [True: 7.86M, False: 7.85M]
  ------------------
  860|  7.86M|            error = atomic_load(&f->task_thread.error);
  861|  7.86M|            const unsigned y = sby + 1 == sbh ? UINT_MAX : (unsigned)(sby + 1) * sbsz;
  ------------------
  |  Branch (861:32): [True: 294k, False: 7.56M]
  ------------------
  862|  7.86M|            assert(c->n_fc > 1);
  ------------------
  |  Branch (862:13): [True: 7.86M, False: 18.4E]
  ------------------
  863|  7.86M|            if (f->sr_cur.p.data[0] /* upon flush, this can be free'ed already */)
  ------------------
  |  Branch (863:17): [True: 7.86M, False: 18.4E]
  ------------------
  864|  7.86M|                atomic_store(&f->sr_cur.progress[0], error ? FRAME_ERROR : y);
  ------------------
  |  Branch (864:17): [True: 6.16M, False: 1.70M]
  ------------------
  865|  7.86M|            atomic_store(&f->frame_thread.entropy_progress,
  ------------------
  |  Branch (865:13): [True: 6.16M, False: 1.70M]
  ------------------
  866|  7.86M|                         error ? TILE_ERROR : sby + 1);
  867|  7.86M|            if (sby + 1 == sbh)
  ------------------
  |  Branch (867:17): [True: 294k, False: 7.56M]
  ------------------
  868|  7.86M|                atomic_store(&f->task_thread.done[1], 1);
  869|  7.86M|            pthread_mutex_lock(&ttd->lock);
  870|  7.86M|            const int num_tasks = atomic_fetch_sub(&f->task_thread.task_counter, 1) - 1;
  871|  7.86M|            if (sby + 1 < sbh && num_tasks) {
  ------------------
  |  Branch (871:17): [True: 7.56M, False: 293k]
  |  Branch (871:34): [True: 7.56M, False: 7.63k]
  ------------------
  872|  7.56M|                reset_task_cur(c, ttd, t->frame_idx);
  873|  7.56M|                continue;
  874|  7.56M|            }
  875|   301k|            if (!num_tasks && atomic_load(&f->task_thread.done[0]) &&
  ------------------
  |  Branch (875:17): [True: 40.3k, False: 260k]
  |  Branch (875:31): [True: 40.3k, False: 0]
  ------------------
  876|   301k|                atomic_load(&f->task_thread.done[1]))
  ------------------
  |  Branch (876:17): [True: 40.3k, False: 0]
  ------------------
  877|  40.3k|            {
  878|  40.3k|                error = atomic_load(&f->task_thread.error);
  879|  40.3k|                dav1d_decode_frame_exit(f, error == 1 ? DAV1D_ERR(EINVAL) :
  ------------------
  |  |   58|  29.1k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (879:44): [True: 29.1k, False: 11.2k]
  ------------------
  880|  40.3k|                                        error ? DAV1D_ERR(ENOMEM) : 0);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (880:41): [True: 0, False: 11.2k]
  ------------------
  881|  40.3k|                f->n_tile_data = 0;
  882|  40.3k|                pthread_cond_signal(&f->task_thread.cond);
  883|  40.3k|            }
  884|   301k|            reset_task_cur(c, ttd, t->frame_idx);
  885|   301k|            continue;
  886|  7.86M|        }
  887|       |    // t->type != DAV1D_TASK_TYPE_ENTROPY_PROGRESS
  888|  15.7M|        atomic_fetch_or(&f->frame_thread.frame_progress[sby >> 5],
  889|  7.85M|                        1U << (sby & 31));
  890|  7.85M|        pthread_mutex_lock(&f->task_thread.lock);
  891|  7.85M|        sby = get_frame_progress(c, f);
  892|  7.85M|        error = atomic_load(&f->task_thread.error);
  893|  7.85M|        const unsigned y = sby + 1 == sbh ? UINT_MAX : (unsigned)(sby + 1) * sbsz;
  ------------------
  |  Branch (893:28): [True: 6.27M, False: 1.58M]
  ------------------
  894|  7.86M|        if (c->n_fc > 1 && f->sr_cur.p.data[0] /* upon flush, this can be free'ed already */)
  ------------------
  |  Branch (894:13): [True: 7.86M, False: 18.4E]
  |  Branch (894:28): [True: 7.86M, False: 12]
  ------------------
  895|  7.86M|            atomic_store(&f->sr_cur.progress[1], error ? FRAME_ERROR : y);
  ------------------
  |  Branch (895:13): [True: 6.18M, False: 1.67M]
  ------------------
  896|  7.85M|        pthread_mutex_unlock(&f->task_thread.lock);
  897|  7.85M|        if (sby + 1 == sbh)
  ------------------
  |  Branch (897:13): [True: 6.27M, False: 1.58M]
  ------------------
  898|  7.85M|            atomic_store(&f->task_thread.done[0], 1);
  899|  7.85M|        pthread_mutex_lock(&ttd->lock);
  900|  7.85M|        const int num_tasks = atomic_fetch_sub(&f->task_thread.task_counter, 1) - 1;
  901|  7.85M|        if (sby + 1 < sbh && num_tasks) {
  ------------------
  |  Branch (901:13): [True: 1.58M, False: 6.27M]
  |  Branch (901:30): [True: 1.58M, False: 4.02k]
  ------------------
  902|  1.58M|            reset_task_cur(c, ttd, t->frame_idx);
  903|  1.58M|            continue;
  904|  1.58M|        }
  905|  6.27M|        if (!num_tasks && atomic_load(&f->task_thread.done[0]) &&
  ------------------
  |  Branch (905:13): [True: 245k, False: 6.03M]
  |  Branch (905:27): [True: 245k, False: 0]
  ------------------
  906|   245k|            (!uses_2pass || atomic_load(&f->task_thread.done[1])))
  ------------------
  |  Branch (906:14): [True: 0, False: 245k]
  |  Branch (906:29): [True: 245k, False: 0]
  ------------------
  907|   245k|        {
  908|   245k|            error = atomic_load(&f->task_thread.error);
  909|   245k|            dav1d_decode_frame_exit(f, error == 1 ? DAV1D_ERR(EINVAL) :
  ------------------
  |  |   58|  85.2k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (909:40): [True: 85.2k, False: 160k]
  ------------------
  910|   245k|                                    error ? DAV1D_ERR(ENOMEM) : 0);
  ------------------
  |  |   58|      0|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (910:37): [True: 0, False: 160k]
  ------------------
  911|   245k|            f->n_tile_data = 0;
  912|   245k|            pthread_cond_signal(&f->task_thread.cond);
  913|   245k|        }
  914|  6.27M|        reset_task_cur(c, ttd, t->frame_idx);
  915|  6.27M|    }
  916|  41.3k|    pthread_mutex_unlock(&ttd->lock);
  917|       |
  918|       |    return NULL;
  919|  37.6k|}
thread_task.c:create_filter_sbrow:
  215|   589k|{
  216|   589k|    const int has_deblock = f->frame_hdr->loopfilter.level_y[0] ||
  ------------------
  |  Branch (216:29): [True: 88.9k, False: 500k]
  ------------------
  217|   500k|                            f->frame_hdr->loopfilter.level_y[1];
  ------------------
  |  Branch (217:29): [True: 3.85k, False: 496k]
  ------------------
  218|   589k|    const int has_cdef = f->seq_hdr->cdef;
  219|   589k|    const int has_resize = f->frame_hdr->width[0] != f->frame_hdr->width[1];
  220|   589k|    const int has_lr = f->lf.restore_planes;
  221|       |
  222|   589k|    Dav1dTask *tasks = f->task_thread.tasks;
  223|   589k|    const int uses_2pass = f->c->n_fc > 1;
  224|   589k|    int num_tasks = f->sbh * (1 + uses_2pass);
  225|   589k|    if (num_tasks > f->task_thread.num_tasks) {
  ------------------
  |  Branch (225:9): [True: 16.0k, False: 573k]
  ------------------
  226|  16.0k|        const size_t size = sizeof(Dav1dTask) * num_tasks;
  227|  16.0k|        tasks = dav1d_realloc(ALLOC_COMMON_CTX, f->task_thread.tasks, size);
  ------------------
  |  |  133|  16.0k|#define dav1d_realloc(type, ptr, sz) realloc(ptr, sz)
  ------------------
  228|  16.0k|        if (!tasks) return -1;
  ------------------
  |  Branch (228:13): [True: 0, False: 16.0k]
  ------------------
  229|  16.0k|        memset(tasks, 0, size);
  230|  16.0k|        f->task_thread.tasks = tasks;
  231|  16.0k|        f->task_thread.num_tasks = num_tasks;
  232|  16.0k|    }
  233|   589k|    tasks += f->sbh * (pass & 1);
  234|       |
  235|   589k|    if (pass & 1) {
  ------------------
  |  Branch (235:9): [True: 294k, False: 294k]
  ------------------
  236|   294k|        f->frame_thread.entropy_progress = 0;
  237|   294k|    } else {
  238|   294k|        const int prog_sz = ((f->sbh + 31) & ~31) >> 5;
  239|   294k|        if (prog_sz > f->frame_thread.prog_sz) {
  ------------------
  |  Branch (239:13): [True: 16.3k, False: 278k]
  ------------------
  240|  16.3k|            atomic_uint *const prog = dav1d_realloc(ALLOC_COMMON_CTX, f->frame_thread.frame_progress,
  ------------------
  |  |  133|  16.3k|#define dav1d_realloc(type, ptr, sz) realloc(ptr, sz)
  ------------------
  241|  16.3k|                                                    2 * prog_sz * sizeof(*prog));
  242|  16.3k|            if (!prog) return -1;
  ------------------
  |  Branch (242:17): [True: 0, False: 16.3k]
  ------------------
  243|  16.3k|            f->frame_thread.frame_progress = prog;
  244|  16.3k|            f->frame_thread.copy_lpf_progress = prog + prog_sz;
  245|  16.3k|        }
  246|   294k|        f->frame_thread.prog_sz = prog_sz;
  247|   294k|        memset(f->frame_thread.frame_progress, 0, prog_sz * sizeof(atomic_uint));
  248|   294k|        memset(f->frame_thread.copy_lpf_progress, 0, prog_sz * sizeof(atomic_uint));
  249|   294k|        atomic_store(&f->frame_thread.deblock_progress, 0);
  250|   294k|    }
  251|   589k|    f->frame_thread.next_tile_row[pass & 1] = 0;
  252|       |
  253|   589k|    Dav1dTask *t = &tasks[0];
  254|   589k|    t->sby = 0;
  255|   589k|    t->recon_progress = 1;
  256|   589k|    t->deblock_progress = 0;
  257|   589k|    t->type = pass == 1 ? DAV1D_TASK_TYPE_ENTROPY_PROGRESS :
  ------------------
  |  Branch (257:15): [True: 295k, False: 294k]
  ------------------
  258|   589k|              has_deblock ? DAV1D_TASK_TYPE_DEBLOCK_COLS :
  ------------------
  |  Branch (258:15): [True: 46.4k, False: 247k]
  ------------------
  259|   294k|              has_cdef || has_lr /* i.e. LR backup */ ? DAV1D_TASK_TYPE_DEBLOCK_ROWS :
  ------------------
  |  Branch (259:15): [True: 177k, False: 69.9k]
  |  Branch (259:27): [True: 719, False: 69.2k]
  ------------------
  260|   247k|              has_resize ? DAV1D_TASK_TYPE_SUPER_RESOLUTION :
  ------------------
  |  Branch (260:15): [True: 1.12k, False: 67.2k]
  ------------------
  261|  68.3k|              DAV1D_TASK_TYPE_RECONSTRUCTION_PROGRESS;
  262|   589k|    t->frame_idx = (int)(f - f->c->fc);
  263|       |
  264|   589k|    *res_t = t;
  265|   589k|    return 0;
  266|   589k|}
thread_task.c:insert_task:
  172|  20.7M|{
  173|  20.7M|    insert_tasks(f, t, t, cond_signal);
  174|  20.7M|}
thread_task.c:insert_tasks:
  118|  20.7M|{
  119|       |    // insert task back into task queue
  120|  20.7M|    Dav1dTask *t_ptr, *prev_t = NULL;
  121|  20.7M|    for (t_ptr = f->task_thread.task_head;
  122|  57.9M|         t_ptr; prev_t = t_ptr, t_ptr = t_ptr->next)
  ------------------
  |  Branch (122:10): [True: 40.2M, False: 17.6M]
  ------------------
  123|  40.2M|    {
  124|       |        // entropy coding precedes other steps
  125|  40.2M|        if (t_ptr->type == DAV1D_TASK_TYPE_TILE_ENTROPY) {
  ------------------
  |  Branch (125:13): [True: 1.54M, False: 38.7M]
  ------------------
  126|  1.54M|            if (first->type > DAV1D_TASK_TYPE_TILE_ENTROPY) continue;
  ------------------
  |  Branch (126:17): [True: 1.19M, False: 348k]
  ------------------
  127|       |            // both are entropy
  128|   348k|            if (first->sby > t_ptr->sby) continue;
  ------------------
  |  Branch (128:17): [True: 173k, False: 175k]
  ------------------
  129|   175k|            if (first->sby < t_ptr->sby) {
  ------------------
  |  Branch (129:17): [True: 39.8k, False: 135k]
  ------------------
  130|  39.8k|                insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
  131|  39.8k|                return;
  132|  39.8k|            }
  133|       |            // same sby
  134|  38.7M|        } else {
  135|  38.7M|            if (first->type == DAV1D_TASK_TYPE_TILE_ENTROPY) {
  ------------------
  |  Branch (135:17): [True: 1.89M, False: 36.8M]
  ------------------
  136|  1.89M|                insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
  137|  1.89M|                return;
  138|  1.89M|            }
  139|  36.8M|            if (first->sby > t_ptr->sby) continue;
  ------------------
  |  Branch (139:17): [True: 26.8M, False: 9.97M]
  ------------------
  140|  9.97M|            if (first->sby < t_ptr->sby) {
  ------------------
  |  Branch (140:17): [True: 1.06M, False: 8.91M]
  ------------------
  141|  1.06M|                insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
  142|  1.06M|                return;
  143|  1.06M|            }
  144|       |            // same sby
  145|  8.91M|            if (first->type > t_ptr->type) continue;
  ------------------
  |  Branch (145:17): [True: 8.82M, False: 93.5k]
  ------------------
  146|  93.5k|            if (first->type < t_ptr->type) {
  ------------------
  |  Branch (146:17): [True: 39.5k, False: 53.9k]
  ------------------
  147|  39.5k|                insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
  148|  39.5k|                return;
  149|  39.5k|            }
  150|       |            // same task type
  151|  93.5k|        }
  152|       |
  153|       |        // sort by tile-id
  154|  40.2M|        assert(first->type == DAV1D_TASK_TYPE_TILE_RECONSTRUCTION ||
  ------------------
  |  Branch (154:9): [True: 53.9k, False: 135k]
  |  Branch (154:9): [True: 135k, False: 0]
  ------------------
  155|   189k|               first->type == DAV1D_TASK_TYPE_TILE_ENTROPY);
  156|   189k|        assert(first->type == t_ptr->type);
  ------------------
  |  Branch (156:9): [True: 189k, False: 0]
  ------------------
  157|   189k|        assert(t_ptr->sby == first->sby);
  ------------------
  |  Branch (157:9): [True: 189k, False: 0]
  ------------------
  158|   189k|        const int p = first->type == DAV1D_TASK_TYPE_TILE_ENTROPY;
  159|   189k|        const int t_tile_idx = (int) (first - f->task_thread.tile_tasks[p]);
  160|   189k|        const int p_tile_idx = (int) (t_ptr - f->task_thread.tile_tasks[p]);
  161|   189k|        assert(t_tile_idx != p_tile_idx);
  ------------------
  |  Branch (161:9): [True: 189k, False: 0]
  ------------------
  162|   189k|        if (t_tile_idx > p_tile_idx) continue;
  ------------------
  |  Branch (162:13): [True: 170k, False: 19.2k]
  ------------------
  163|  19.2k|        insert_tasks_between(f, first, last, prev_t, t_ptr, cond_signal);
  164|  19.2k|        return;
  165|   189k|    }
  166|       |    // append at the end
  167|  17.6M|    insert_tasks_between(f, first, last, prev_t, NULL, cond_signal);
  168|  17.6M|}
thread_task.c:insert_tasks_between:
  102|  20.7M|{
  103|  20.7M|    struct TaskThreadData *const ttd = f->task_thread.ttd;
  104|  20.7M|    if (atomic_load(f->c->flush)) return;
  ------------------
  |  Branch (104:9): [True: 47, False: 20.7M]
  ------------------
  105|  20.7M|    assert(!a || a->next == b);
  ------------------
  |  Branch (105:5): [True: 2.58M, False: 18.1M]
  |  Branch (105:5): [True: 18.1M, False: 0]
  ------------------
  106|  20.7M|    if (!a) f->task_thread.task_head = first;
  ------------------
  |  Branch (106:9): [True: 2.58M, False: 18.1M]
  ------------------
  107|  18.1M|    else a->next = first;
  108|  20.7M|    if (!b) f->task_thread.task_tail = last;
  ------------------
  |  Branch (108:9): [True: 17.6M, False: 3.05M]
  ------------------
  109|  20.7M|    last->next = b;
  110|  20.7M|    reset_task_cur(f->c, ttd, first->frame_idx);
  111|  20.7M|    if (cond_signal && !atomic_fetch_or(&ttd->cond_signaled, 1))
  ------------------
  |  Branch (111:9): [True: 356k, False: 20.3M]
  |  Branch (111:24): [True: 118k, False: 237k]
  ------------------
  112|   118k|        pthread_cond_signal(&ttd->cond);
  113|  20.7M|}
thread_task.c:merge_pending:
  206|  22.5M|static inline int merge_pending(const Dav1dContext *const c) {
  207|  22.5M|    int res = 0;
  208|   112M|    for (unsigned i = 0; i < c->n_fc; i++)
  ------------------
  |  Branch (208:26): [True: 90.2M, False: 22.5M]
  ------------------
  209|  90.2M|        res |= merge_pending_frame(&c->fc[i]);
  210|  22.5M|    return res;
  211|  22.5M|}
thread_task.c:delayed_fg_task:
  473|  16.2k|{
  474|  16.2k|    const Dav1dPicture *const in = ttd->delayed_fg.in;
  475|  16.2k|    Dav1dPicture *const out = ttd->delayed_fg.out;
  476|  16.2k|#if CONFIG_16BPC
  477|  16.2k|    int off;
  478|  16.2k|    if (out->p.bpc != 8)
  ------------------
  |  Branch (478:9): [True: 3.81k, False: 12.4k]
  ------------------
  479|  3.81k|        off = (out->p.bpc >> 1) - 4;
  480|  16.2k|#endif
  481|  16.2k|    switch (ttd->delayed_fg.type) {
  482|  10.1k|    case DAV1D_TASK_TYPE_FG_PREP:
  ------------------
  |  Branch (482:5): [True: 10.1k, False: 6.16k]
  ------------------
  483|  10.1k|        ttd->delayed_fg.exec = 0;
  484|  10.1k|        if (atomic_load(&ttd->cond_signaled))
  ------------------
  |  Branch (484:13): [True: 3.08k, False: 7.03k]
  ------------------
  485|  3.08k|            pthread_cond_signal(&ttd->cond);
  486|  10.1k|        pthread_mutex_unlock(&ttd->lock);
  487|  10.1k|        switch (out->p.bpc) {
  488|      0|#if CONFIG_8BPC
  489|  7.70k|        case 8:
  ------------------
  |  Branch (489:9): [True: 7.70k, False: 2.40k]
  ------------------
  490|  7.70k|            dav1d_prep_grain_8bpc(&c->dsp[0].fg, out, in,
  491|  7.70k|                                  ttd->delayed_fg.scaling_8bpc,
  492|  7.70k|                                  ttd->delayed_fg.grain_lut_8bpc);
  493|  7.70k|            break;
  494|      0|#endif
  495|      0|#if CONFIG_16BPC
  496|  2.06k|        case 10:
  ------------------
  |  Branch (496:9): [True: 2.06k, False: 8.05k]
  ------------------
  497|  2.40k|        case 12:
  ------------------
  |  Branch (497:9): [True: 348, False: 9.76k]
  ------------------
  498|  2.40k|            dav1d_prep_grain_16bpc(&c->dsp[off].fg, out, in,
  499|  2.40k|                                   ttd->delayed_fg.scaling_16bpc,
  500|  2.40k|                                   ttd->delayed_fg.grain_lut_16bpc);
  501|  2.40k|            break;
  502|      0|#endif
  503|      0|        default: abort();
  ------------------
  |  Branch (503:9): [True: 0, False: 10.1k]
  ------------------
  504|  10.1k|        }
  505|  10.1k|        ttd->delayed_fg.type = DAV1D_TASK_TYPE_FG_APPLY;
  506|  10.1k|        pthread_mutex_lock(&ttd->lock);
  507|  10.1k|        ttd->delayed_fg.exec = 1;
  508|       |        // fall-through
  509|  16.2k|    case DAV1D_TASK_TYPE_FG_APPLY:;
  ------------------
  |  Branch (509:5): [True: 6.16k, False: 10.1k]
  ------------------
  510|  16.2k|        int row = atomic_fetch_add(&ttd->delayed_fg.progress[0], 1);
  511|  16.2k|        pthread_mutex_unlock(&ttd->lock);
  512|  16.2k|        int progmax = (out->p.h + FG_BLOCK_SIZE - 1) / FG_BLOCK_SIZE;
  ------------------
  |  |   37|  16.2k|#define FG_BLOCK_SIZE 32
  ------------------
                      int progmax = (out->p.h + FG_BLOCK_SIZE - 1) / FG_BLOCK_SIZE;
  ------------------
  |  |   37|  16.2k|#define FG_BLOCK_SIZE 32
  ------------------
  513|  55.2k|        while (row < progmax) {
  ------------------
  |  Branch (513:16): [True: 38.9k, False: 16.2k]
  ------------------
  514|  38.9k|            if (row + 1 < progmax)
  ------------------
  |  Branch (514:17): [True: 28.8k, False: 10.1k]
  ------------------
  515|  28.8k|                pthread_cond_signal(&ttd->cond);
  516|  10.1k|            else {
  517|  10.1k|                pthread_mutex_lock(&ttd->lock);
  518|  10.1k|                ttd->delayed_fg.exec = 0;
  519|  10.1k|                pthread_mutex_unlock(&ttd->lock);
  520|  10.1k|            }
  521|  38.9k|            switch (out->p.bpc) {
  522|      0|#if CONFIG_8BPC
  523|  29.8k|            case 8:
  ------------------
  |  Branch (523:13): [True: 29.8k, False: 9.16k]
  ------------------
  524|  29.8k|                dav1d_apply_grain_row_8bpc(&c->dsp[0].fg, out, in,
  525|  29.8k|                                           ttd->delayed_fg.scaling_8bpc,
  526|  29.8k|                                           ttd->delayed_fg.grain_lut_8bpc, row);
  527|  29.8k|                break;
  528|      0|#endif
  529|      0|#if CONFIG_16BPC
  530|  7.79k|            case 10:
  ------------------
  |  Branch (530:13): [True: 7.79k, False: 31.2k]
  ------------------
  531|  9.16k|            case 12:
  ------------------
  |  Branch (531:13): [True: 1.37k, False: 37.6k]
  ------------------
  532|  9.16k|                dav1d_apply_grain_row_16bpc(&c->dsp[off].fg, out, in,
  533|  9.16k|                                            ttd->delayed_fg.scaling_16bpc,
  534|  9.16k|                                            ttd->delayed_fg.grain_lut_16bpc, row);
  535|  9.16k|                break;
  536|      0|#endif
  537|      0|            default: abort();
  ------------------
  |  Branch (537:13): [True: 0, False: 38.9k]
  ------------------
  538|  38.9k|            }
  539|  38.9k|            row = atomic_fetch_add(&ttd->delayed_fg.progress[0], 1);
  540|  38.9k|            atomic_fetch_add(&ttd->delayed_fg.progress[1], 1);
  541|  38.9k|        }
  542|  16.2k|        pthread_mutex_lock(&ttd->lock);
  543|  16.2k|        ttd->delayed_fg.exec = 0;
  544|  16.2k|        int done = atomic_fetch_add(&ttd->delayed_fg.progress[1], 1) + 1;
  545|  16.2k|        progmax = atomic_load(&ttd->delayed_fg.progress[0]);
  546|       |        // signal for completion only once the last runner reaches this
  547|  16.2k|        if (done >= progmax) {
  ------------------
  |  Branch (547:13): [True: 10.1k, False: 6.08k]
  ------------------
  548|  10.1k|            ttd->delayed_fg.finished = 1;
  549|  10.1k|            pthread_cond_signal(&ttd->delayed_fg.cond);
  550|  10.1k|        }
  551|  16.2k|        break;
  552|      0|    default: abort();
  ------------------
  |  Branch (552:5): [True: 0, False: 16.2k]
  ------------------
  553|  16.2k|    }
  554|  16.2k|}
thread_task.c:merge_pending_frame:
  188|   113M|static inline int merge_pending_frame(Dav1dFrameContext *const f) {
  189|   113M|    int const merge = atomic_load(&f->task_thread.pending_tasks.merge);
  190|   113M|    if (merge) {
  ------------------
  |  Branch (190:9): [True: 4.20M, False: 109M]
  ------------------
  191|  4.20M|        pthread_mutex_lock(&f->task_thread.pending_tasks.lock);
  192|  4.20M|        Dav1dTask *t = f->task_thread.pending_tasks.head;
  193|  4.20M|        f->task_thread.pending_tasks.head = NULL;
  194|  4.20M|        f->task_thread.pending_tasks.tail = NULL;
  195|  4.20M|        atomic_store(&f->task_thread.pending_tasks.merge, 0);
  196|  4.20M|        pthread_mutex_unlock(&f->task_thread.pending_tasks.lock);
  197|  9.44M|        while (t) {
  ------------------
  |  Branch (197:16): [True: 5.23M, False: 4.20M]
  ------------------
  198|  5.23M|            Dav1dTask *const tmp = t->next;
  199|  5.23M|            insert_task(f, t, 0);
  200|  5.23M|            t = tmp;
  201|  5.23M|        }
  202|  4.20M|    }
  203|   113M|    return merge;
  204|   113M|}
thread_task.c:check_tile:
  395|  20.6M|{
  396|  20.6M|    const int tp = t->type == DAV1D_TASK_TYPE_TILE_ENTROPY;
  397|  20.6M|    const int tile_idx = (int)(t - f->task_thread.tile_tasks[tp]);
  398|  20.6M|    Dav1dTileState *const ts = &f->ts[tile_idx];
  399|  20.6M|    const int p1 = atomic_load(&ts->progress[tp]);
  400|  20.6M|    if (p1 < t->sby) return 1;
  ------------------
  |  Branch (400:9): [True: 3.83M, False: 16.8M]
  ------------------
  401|  16.8M|    int error = p1 == TILE_ERROR;
  ------------------
  |  |   36|  16.8M|#define TILE_ERROR (INT_MAX - 1)
  ------------------
  402|  16.8M|    error |= atomic_fetch_or(&f->task_thread.error, error);
  403|  16.8M|    if (!error && frame_mt && !tp) {
  ------------------
  |  Branch (403:9): [True: 4.51M, False: 12.3M]
  |  Branch (403:19): [True: 4.51M, False: 0]
  |  Branch (403:31): [True: 2.32M, False: 2.18M]
  ------------------
  404|  2.32M|        const int p2 = atomic_load(&ts->progress[1]);
  405|  2.32M|        if (p2 <= t->sby) return 1;
  ------------------
  |  Branch (405:13): [True: 244k, False: 2.08M]
  ------------------
  406|  2.08M|        error = p2 == TILE_ERROR;
  ------------------
  |  |   36|  2.08M|#define TILE_ERROR (INT_MAX - 1)
  ------------------
  407|  2.08M|        error |= atomic_fetch_or(&f->task_thread.error, error);
  408|  2.08M|    }
  409|  16.5M|    if (!error && frame_mt && !IS_KEY_OR_INTRA(f->frame_hdr)) {
  ------------------
  |  |   43|  4.26M|    (!IS_INTER_OR_SWITCH(frame_header))
  |  |  ------------------
  |  |  |  |   36|  4.26M|    ((frame_header)->frame_type & 1)
  |  |  ------------------
  ------------------
  |  Branch (409:9): [True: 4.26M, False: 12.3M]
  |  Branch (409:19): [True: 4.26M, False: 0]
  |  Branch (409:31): [True: 3.19M, False: 1.06M]
  ------------------
  410|       |        // check reference state
  411|  3.19M|        const Dav1dThreadPicture *p = &f->sr_cur;
  412|  3.19M|        const int ss_ver = p->p.p.layout == DAV1D_PIXEL_LAYOUT_I420;
  413|  3.19M|        const unsigned p_b = (t->sby + 1) << (f->sb_shift + 2);
  414|  3.19M|        const int tile_sby = t->sby - (ts->tiling.row_start >> f->sb_shift);
  415|  3.19M|        const int (*const lowest_px)[2] = ts->lowest_pixel[tile_sby];
  416|  24.7M|        for (int n = t->deps_skip; n < 7; n++, t->deps_skip++) {
  ------------------
  |  Branch (416:36): [True: 21.7M, False: 3.08M]
  ------------------
  417|  21.7M|            unsigned lowest;
  418|  21.7M|            if (tp) {
  ------------------
  |  Branch (418:17): [True: 11.0M, False: 10.6M]
  ------------------
  419|       |                // if temporal mv refs are disabled, we only need this
  420|       |                // for the primary ref; if segmentation is disabled, we
  421|       |                // don't even need that
  422|  11.0M|                lowest = p_b;
  423|  11.0M|            } else {
  424|       |                // +8 is postfilter-induced delay
  425|  10.6M|                const int y = lowest_px[n][0] == INT_MIN ? INT_MIN :
  ------------------
  |  Branch (425:31): [True: 9.02M, False: 1.63M]
  ------------------
  426|  10.6M|                              lowest_px[n][0] + 8;
  427|  10.6M|                const int uv = lowest_px[n][1] == INT_MIN ? INT_MIN :
  ------------------
  |  Branch (427:32): [True: 10.5M, False: 157k]
  ------------------
  428|  10.6M|                               lowest_px[n][1] * (1 << ss_ver) + 8;
  429|  10.6M|                const int max = imax(y, uv);
  430|  10.6M|                if (max == INT_MIN) continue;
  ------------------
  |  Branch (430:21): [True: 9.02M, False: 1.63M]
  ------------------
  431|  1.63M|                lowest = iclip(max, 1, f->refp[n].p.p.h);
  432|  1.63M|            }
  433|  12.6M|            const unsigned p3 = atomic_load(&f->refp[n].progress[!tp]);
  434|  12.6M|            if (p3 < lowest) return 1;
  ------------------
  |  Branch (434:17): [True: 115k, False: 12.5M]
  ------------------
  435|  12.6M|            atomic_fetch_or(&f->task_thread.error, p3 == FRAME_ERROR);
  436|  12.5M|        }
  437|  3.19M|    }
  438|  16.4M|    return 0;
  439|  16.5M|}
thread_task.c:reset_task_cur:
   50|  38.9M|{
   51|  38.9M|    const unsigned first = atomic_load(&ttd->first);
   52|  38.9M|    unsigned reset_frame_idx = atomic_exchange(&ttd->reset_task_cur, UINT_MAX);
   53|  38.9M|    if (reset_frame_idx < first) {
  ------------------
  |  Branch (53:9): [True: 0, False: 38.9M]
  ------------------
   54|      0|        if (frame_idx == UINT_MAX) return 0;
  ------------------
  |  Branch (54:13): [True: 0, False: 0]
  ------------------
   55|      0|        reset_frame_idx = UINT_MAX;
   56|      0|    }
   57|  38.9M|    if (!ttd->cur && c->fc[first].task_thread.task_cur_prev == NULL)
  ------------------
  |  Branch (57:9): [True: 26.5M, False: 12.4M]
  |  Branch (57:22): [True: 24.3M, False: 2.22M]
  ------------------
   58|  24.3M|        return 0;
   59|  14.6M|    if (reset_frame_idx != UINT_MAX) {
  ------------------
  |  Branch (59:9): [True: 1.98M, False: 12.6M]
  ------------------
   60|  1.98M|        if (frame_idx == UINT_MAX) {
  ------------------
  |  Branch (60:13): [True: 37.7k, False: 1.94M]
  ------------------
   61|  37.7k|            if (reset_frame_idx > first + ttd->cur)
  ------------------
  |  Branch (61:17): [True: 361, False: 37.3k]
  ------------------
   62|    361|                return 0;
   63|  37.3k|            ttd->cur = reset_frame_idx - first;
   64|  37.3k|            goto cur_found;
   65|  37.7k|        }
   66|  12.6M|    } else if (frame_idx == UINT_MAX)
  ------------------
  |  Branch (66:16): [True: 1.47M, False: 11.2M]
  ------------------
   67|  1.47M|        return 0;
   68|  13.1M|    if (frame_idx < first) frame_idx += c->n_fc;
  ------------------
  |  Branch (68:9): [True: 3.50M, False: 9.64M]
  ------------------
   69|  13.1M|    const unsigned min_frame_idx = umin(reset_frame_idx, frame_idx);
   70|  13.1M|    const unsigned cur_frame_idx = first + ttd->cur;
   71|  13.1M|    if (ttd->cur < c->n_fc && cur_frame_idx < min_frame_idx)
  ------------------
  |  Branch (71:9): [True: 12.5M, False: 644k]
  |  Branch (71:31): [True: 798k, False: 11.7M]
  ------------------
   72|   798k|        return 0;
   73|  13.0M|    for (ttd->cur = min_frame_idx - first; ttd->cur < c->n_fc; ttd->cur++)
  ------------------
  |  Branch (73:44): [True: 12.9M, False: 136k]
  ------------------
   74|  12.9M|        if (c->fc[(first + ttd->cur) % c->n_fc].task_thread.task_head)
  ------------------
  |  Branch (74:13): [True: 12.2M, False: 691k]
  ------------------
   75|  12.2M|            break;
   76|  12.3M|cur_found:
   77|  48.4M|    for (unsigned i = ttd->cur; i < c->n_fc; i++)
  ------------------
  |  Branch (77:33): [True: 36.0M, False: 12.3M]
  ------------------
   78|  36.0M|        c->fc[(first + i) % c->n_fc].task_thread.task_cur_prev = NULL;
   79|  12.3M|    return 1;
   80|  12.3M|}
thread_task.c:abort_frame:
  459|  60.1k|static inline void abort_frame(Dav1dFrameContext *const f, const int error) {
  460|  60.1k|    atomic_store(&f->task_thread.error, error == DAV1D_ERR(EINVAL) ? 1 : -1);
  ------------------
  |  Branch (460:5): [True: 46.5k, False: 13.5k]
  ------------------
  461|  60.1k|    atomic_store(&f->task_thread.task_counter, 0);
  462|  60.1k|    atomic_store(&f->task_thread.done[0], 1);
  463|  60.1k|    atomic_store(&f->task_thread.done[1], 1);
  464|  60.1k|    atomic_store(&f->sr_cur.progress[0], FRAME_ERROR);
  465|       |    atomic_store(&f->sr_cur.progress[1], FRAME_ERROR);
  466|  60.1k|    dav1d_decode_frame_exit(f, error);
  467|  60.1k|    f->n_tile_data = 0;
  468|  60.1k|    pthread_cond_signal(&f->task_thread.cond);
  469|  60.1k|}
thread_task.c:add_pending:
  176|  4.01M|static inline void add_pending(Dav1dFrameContext *const f, Dav1dTask *const t) {
  177|  4.01M|    pthread_mutex_lock(&f->task_thread.pending_tasks.lock);
  178|  4.01M|    t->next = NULL;
  179|  4.01M|    if (!f->task_thread.pending_tasks.head)
  ------------------
  |  Branch (179:9): [True: 3.90M, False: 109k]
  ------------------
  180|  3.90M|        f->task_thread.pending_tasks.head = t;
  181|   109k|    else
  182|   109k|        f->task_thread.pending_tasks.tail->next = t;
  183|  4.01M|    f->task_thread.pending_tasks.tail = t;
  184|       |    atomic_store(&f->task_thread.pending_tasks.merge, 1);
  185|  4.01M|    pthread_mutex_unlock(&f->task_thread.pending_tasks.lock);
  186|  4.01M|}
thread_task.c:reset_task_cur_async:
   84|  20.2M|{
   85|  20.2M|    const unsigned first = atomic_load(&ttd->first);
   86|  20.2M|    if (frame_idx < first) frame_idx += n_frames;
  ------------------
  |  Branch (86:9): [True: 3.56M, False: 16.6M]
  ------------------
   87|  20.2M|    unsigned last_idx = frame_idx;
   88|  20.4M|    do {
   89|  20.4M|        frame_idx = last_idx;
   90|  20.4M|        last_idx = atomic_exchange(&ttd->reset_task_cur, frame_idx);
   91|  20.4M|    } while (last_idx < frame_idx);
  ------------------
  |  Branch (91:14): [True: 151k, False: 20.2M]
  ------------------
   92|  20.2M|    if (frame_idx == first && atomic_load(&ttd->first) != first) {
  ------------------
  |  Branch (92:9): [True: 7.61M, False: 12.6M]
  |  Branch (92:31): [True: 0, False: 7.61M]
  ------------------
   93|      0|        unsigned expected = frame_idx;
   94|       |        atomic_compare_exchange_strong(&ttd->reset_task_cur, &expected, UINT_MAX);
   95|      0|    }
   96|  20.2M|}
thread_task.c:ensure_progress:
  378|  2.03M|{
  379|       |    // deblock_rows (non-LR portion) depends on deblock of previous sbrow,
  380|       |    // so ensure that completed. if not, re-add to task-queue; else, fall-through
  381|  2.03M|    int p1 = atomic_load(state);
  382|  2.03M|    if (p1 < t->sby) {
  ------------------
  |  Branch (382:9): [True: 18.4k, False: 2.01M]
  ------------------
  383|  18.4k|        t->type = type;
  384|  18.4k|        t->recon_progress = t->deblock_progress = 0;
  385|  18.4k|        *target = t->sby;
  386|  18.4k|        add_pending(f, t);
  387|  18.4k|        pthread_mutex_lock(&ttd->lock);
  388|  18.4k|        return 1;
  389|  18.4k|    }
  390|  2.01M|    return 0;
  391|  2.03M|}
thread_task.c:get_frame_progress:
  443|  7.86M|{
  444|  7.86M|    unsigned frame_prog = c->n_fc > 1 ? atomic_load(&f->sr_cur.progress[1]) : 0;
  ------------------
  |  Branch (444:27): [True: 7.86M, False: 16]
  ------------------
  445|  7.86M|    if (frame_prog >= FRAME_ERROR)
  ------------------
  |  |   35|  7.86M|#define FRAME_ERROR (UINT_MAX - 1)
  ------------------
  |  Branch (445:9): [True: 6.06M, False: 1.79M]
  ------------------
  446|  6.06M|        return f->sbh - 1;
  447|  1.79M|    int idx = frame_prog >> (f->sb_shift + 7);
  448|  1.79M|    int prog;
  449|  1.84M|    do {
  450|  1.84M|        atomic_uint *state = &f->frame_thread.frame_progress[idx];
  451|  1.84M|        const unsigned val = ~atomic_load(state);
  452|  1.84M|        prog = val ? ctz(val) : 32;
  ------------------
  |  Branch (452:16): [True: 1.79M, False: 46.3k]
  ------------------
  453|  1.84M|        if (prog != 32) break;
  ------------------
  |  Branch (453:13): [True: 1.79M, False: 46.3k]
  ------------------
  454|  46.3k|        prog = 0;
  455|  46.3k|    } while (++idx < f->frame_thread.prog_sz);
  ------------------
  |  Branch (455:14): [True: 43.6k, False: 2.77k]
  ------------------
  456|  1.79M|    return ((idx << 5) | prog) - 1;
  457|  7.86M|}

dav1d_get_shear_params:
   80|   312k|int dav1d_get_shear_params(Dav1dWarpedMotionParams *const wm) {
   81|   312k|    const int32_t *const mat = wm->matrix;
   82|       |
   83|   312k|    if (mat[2] <= 0) return 1;
  ------------------
  |  Branch (83:9): [True: 0, False: 312k]
  ------------------
   84|       |
   85|   312k|    wm->u.p.alpha = iclip_wmp(mat[2] - 0x10000);
   86|   312k|    wm->u.p.beta = iclip_wmp(mat[3]);
   87|       |
   88|   312k|    int shift;
   89|   312k|    const int y = apply_sign(resolve_divisor_32(abs(mat[2]), &shift), mat[2]);
   90|   312k|    const int64_t v1 = ((int64_t) mat[4] * 0x10000) * y;
   91|   312k|    const int rnd = (1 << shift) >> 1;
   92|   312k|    wm->u.p.gamma = iclip_wmp(apply_sign64((int) ((llabs(v1) + rnd) >> shift), v1));
   93|   312k|    const int64_t v2 = ((int64_t) mat[3] * mat[4]) * y;
   94|   312k|    wm->u.p.delta = iclip_wmp(mat[5] -
   95|   312k|                          apply_sign64((int) ((llabs(v2) + rnd) >> shift), v2) -
   96|   312k|                          0x10000);
   97|       |
   98|   312k|    return (4 * abs(wm->u.p.alpha) + 7 * abs(wm->u.p.beta) >= 0x10000) ||
  ------------------
  |  Branch (98:12): [True: 8.31k, False: 304k]
  ------------------
   99|   304k|           (4 * abs(wm->u.p.gamma) + 4 * abs(wm->u.p.delta) >= 0x10000);
  ------------------
  |  Branch (99:12): [True: 1.71k, False: 302k]
  ------------------
  100|   312k|}
dav1d_set_affine_mv2d:
  136|   124k|{
  137|   124k|    int32_t *const mat = wm->matrix;
  138|   124k|    const int rsuy = 2 * bh4 - 1;
  139|   124k|    const int rsux = 2 * bw4 - 1;
  140|   124k|    const int isuy = by4 * 4 + rsuy;
  141|   124k|    const int isux = bx4 * 4 + rsux;
  142|       |
  143|   124k|    mat[0] = iclip(mv.x * 0x2000 - (isux * (mat[2] - 0x10000) + isuy * mat[3]),
  144|   124k|                   -0x800000, 0x7fffff);
  145|   124k|    mat[1] = iclip(mv.y * 0x2000 - (isux * mat[4] + isuy * (mat[5] - 0x10000)),
  146|   124k|                   -0x800000, 0x7fffff);
  147|   124k|}
dav1d_find_affine_int:
  153|   169k|{
  154|   169k|    int32_t *const mat = wm->matrix;
  155|   169k|    int a[2][2] = { { 0, 0 }, { 0, 0 } };
  156|   169k|    int bx[2] = { 0, 0 };
  157|   169k|    int by[2] = { 0, 0 };
  158|   169k|    const int rsuy = 2 * bh4 - 1;
  159|   169k|    const int rsux = 2 * bw4 - 1;
  160|   169k|    const int suy = rsuy * 8;
  161|   169k|    const int sux = rsux * 8;
  162|   169k|    const int duy = suy + mv.y;
  163|   169k|    const int dux = sux + mv.x;
  164|   169k|    const int isuy = by4 * 4 + rsuy;
  165|   169k|    const int isux = bx4 * 4 + rsux;
  166|       |
  167|   493k|    for (int i = 0; i < np; i++) {
  ------------------
  |  Branch (167:21): [True: 323k, False: 169k]
  ------------------
  168|   323k|        const int dx = pts[i][1][0] - dux;
  169|   323k|        const int dy = pts[i][1][1] - duy;
  170|   323k|        const int sx = pts[i][0][0] - sux;
  171|   323k|        const int sy = pts[i][0][1] - suy;
  172|   323k|        if (abs(sx - dx) < 256 && abs(sy - dy) < 256) {
  ------------------
  |  Branch (172:13): [True: 319k, False: 4.20k]
  |  Branch (172:35): [True: 315k, False: 4.04k]
  ------------------
  173|   315k|            a[0][0] += ((sx * sx) >> 2) + sx * 2 + 8;
  174|   315k|            a[0][1] += ((sx * sy) >> 2) + sx + sy + 4;
  175|   315k|            a[1][1] += ((sy * sy) >> 2) + sy * 2 + 8;
  176|   315k|            bx[0] += ((sx * dx) >> 2) + sx + dx + 8;
  177|   315k|            bx[1] += ((sy * dx) >> 2) + sy + dx + 4;
  178|   315k|            by[0] += ((sx * dy) >> 2) + sx + dy + 4;
  179|   315k|            by[1] += ((sy * dy) >> 2) + sy + dy + 8;
  180|   315k|        }
  181|   323k|    }
  182|       |
  183|       |    // compute determinant of a
  184|   169k|    const int64_t det = (int64_t) a[0][0] * a[1][1] - (int64_t) a[0][1] * a[0][1];
  185|   169k|    if (det == 0) return 1;
  ------------------
  |  Branch (185:9): [True: 8.25k, False: 161k]
  ------------------
  186|   161k|    int shift, idet = apply_sign64(resolve_divisor_64(llabs(det), &shift), det);
  187|   161k|    shift -= 16;
  188|   161k|    if (shift < 0) {
  ------------------
  |  Branch (188:9): [True: 0, False: 161k]
  ------------------
  189|      0|        idet <<= -shift;
  190|      0|        shift = 0;
  191|      0|    }
  192|       |
  193|       |    // solve the least-squares
  194|   161k|    mat[2] = get_mult_shift_diag((int64_t) a[1][1] * bx[0] -
  195|   161k|                                 (int64_t) a[0][1] * bx[1], idet, shift);
  196|   161k|    mat[3] = get_mult_shift_ndiag((int64_t) a[0][0] * bx[1] -
  197|   161k|                                  (int64_t) a[0][1] * bx[0], idet, shift);
  198|   161k|    mat[4] = get_mult_shift_ndiag((int64_t) a[1][1] * by[0] -
  199|   161k|                                  (int64_t) a[0][1] * by[1], idet, shift);
  200|   161k|    mat[5] = get_mult_shift_diag((int64_t) a[0][0] * by[1] -
  201|   161k|                                 (int64_t) a[0][1] * by[0], idet, shift);
  202|       |
  203|   161k|    mat[0] = iclip(mv.x * 0x2000 - (isux * (mat[2] - 0x10000) + isuy * mat[3]),
  204|   161k|                   -0x800000, 0x7fffff);
  205|   161k|    mat[1] = iclip(mv.y * 0x2000 - (isux * mat[4] + isuy * (mat[5] - 0x10000)),
  206|   161k|                   -0x800000, 0x7fffff);
  207|       |
  208|   161k|    return 0;
  209|   169k|}
warpmv.c:iclip_wmp:
   63|  1.24M|static inline int iclip_wmp(const int v) {
   64|  1.24M|    const int cv = iclip(v, INT16_MIN, INT16_MAX);
   65|       |
   66|  1.24M|    return apply_sign((abs(cv) + 32) >> 6, cv) * (1 << 6);
   67|  1.24M|}
warpmv.c:resolve_divisor_32:
   69|   312k|static inline int resolve_divisor_32(const unsigned d, int *const shift) {
   70|   312k|    *shift = ulog2(d);
   71|   312k|    const int e = d - (1 << *shift);
   72|   312k|    const int f = *shift > 8 ? (e + (1 << (*shift - 9))) >> (*shift - 8) :
  ------------------
  |  Branch (72:19): [True: 312k, False: 9]
  ------------------
   73|   312k|                               e << (8 - *shift);
   74|   312k|    assert(f <= 256);
  ------------------
  |  Branch (74:5): [True: 311k, False: 26]
  ------------------
   75|   311k|    *shift += 14;
   76|       |    // Use f as lookup into the precomputed table of multipliers
   77|   311k|    return div_lut[f];
   78|   312k|}
warpmv.c:resolve_divisor_64:
  102|   161k|static int resolve_divisor_64(const uint64_t d, int *const shift) {
  103|   161k|    *shift = u64log2(d);
  104|   161k|    const int64_t e = d - (1LL << *shift);
  105|   161k|    const int64_t f = *shift > 8 ? (e + (1LL << (*shift - 9))) >> (*shift - 8) :
  ------------------
  |  Branch (105:23): [True: 161k, False: 18.4E]
  ------------------
  106|  18.4E|                                   e << (8 - *shift);
  107|   161k|    assert(f <= 256);
  ------------------
  |  Branch (107:5): [True: 161k, False: 18.4E]
  ------------------
  108|   161k|    *shift += 14;
  109|       |    // Use f as lookup into the precomputed table of multipliers
  110|   161k|    return div_lut[f];
  111|   161k|}
warpmv.c:get_mult_shift_diag:
  125|   322k|{
  126|   322k|    const int64_t v1 = px * idet;
  127|   322k|    const int v2 = apply_sign64((int) ((llabs(v1) +
  128|   322k|                                        ((1LL << shift) >> 1)) >> shift),
  129|   322k|                                v1);
  130|   322k|    return iclip(v2, 0xe001, 0x11fff);
  131|   322k|}
warpmv.c:get_mult_shift_ndiag:
  115|   322k|{
  116|   322k|    const int64_t v1 = px * idet;
  117|   322k|    const int v2 = apply_sign64((int) ((llabs(v1) +
  118|   322k|                                        ((1LL << shift) >> 1)) >> shift),
  119|   322k|                                v1);
  120|   322k|    return iclip(v2, -0x1fff, 0x1fff);
  121|   322k|}

dav1d_init_ii_wedge_masks:
  207|      1|COLD void dav1d_init_ii_wedge_masks(void) {
  208|       |    // This function is guaranteed to be called only once
  209|       |
  210|      1|    enum WedgeMasterLineType {
  211|      1|        WEDGE_MASTER_LINE_ODD,
  212|      1|        WEDGE_MASTER_LINE_EVEN,
  213|      1|        WEDGE_MASTER_LINE_VERT,
  214|      1|        N_WEDGE_MASTER_LINES,
  215|      1|    };
  216|      1|    static const uint8_t wedge_master_border[N_WEDGE_MASTER_LINES][8] = {
  217|      1|        [WEDGE_MASTER_LINE_ODD]  = {  1,  2,  6, 18, 37, 53, 60, 63 },
  218|      1|        [WEDGE_MASTER_LINE_EVEN] = {  1,  4, 11, 27, 46, 58, 62, 63 },
  219|      1|        [WEDGE_MASTER_LINE_VERT] = {  0,  2,  7, 21, 43, 57, 62, 64 },
  220|      1|    };
  221|      1|    uint8_t master[6][64 * 64];
  222|       |
  223|       |    // create master templates
  224|     65|    for (int y = 0, off = 0; y < 64; y++, off += 64)
  ------------------
  |  Branch (224:30): [True: 64, False: 1]
  ------------------
  225|     64|        insert_border(&master[WEDGE_VERTICAL][off],
  226|     64|                      wedge_master_border[WEDGE_MASTER_LINE_VERT], 32);
  227|     33|    for (int y = 0, off = 0, ctr = 48; y < 64; y += 2, off += 128, ctr--)
  ------------------
  |  Branch (227:40): [True: 32, False: 1]
  ------------------
  228|     32|    {
  229|     32|        insert_border(&master[WEDGE_OBLIQUE63][off],
  230|     32|                      wedge_master_border[WEDGE_MASTER_LINE_EVEN], ctr);
  231|     32|        insert_border(&master[WEDGE_OBLIQUE63][off + 64],
  232|     32|                      wedge_master_border[WEDGE_MASTER_LINE_ODD], ctr - 1);
  233|     32|    }
  234|       |
  235|      1|    transpose(master[WEDGE_OBLIQUE27], master[WEDGE_OBLIQUE63]);
  236|      1|    transpose(master[WEDGE_HORIZONTAL], master[WEDGE_VERTICAL]);
  237|      1|    hflip(master[WEDGE_OBLIQUE117], master[WEDGE_OBLIQUE63]);
  238|      1|    hflip(master[WEDGE_OBLIQUE153], master[WEDGE_OBLIQUE27]);
  239|       |
  240|      1|#define fill(w, h, sz_422, sz_420, hvsw, signs) \
  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  242|      1|                master, wedge_codebook_16_##hvsw, \
  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  246|       |
  247|      1|    fill(32, 32, 16x32, 16x16, heqw, 0x7bfb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  248|      1|    fill(32, 16, 16x16, 16x8,  hltw, 0x7beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  249|      1|    fill(32,  8, 16x8,  16x4,  hltw, 0x6beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  250|      1|    fill(16, 32,  8x32,  8x16, hgtw, 0x7beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  251|      1|    fill(16, 16,  8x16,  8x8,  heqw, 0x7bfb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  252|      1|    fill(16,  8,  8x8,   8x4,  hltw, 0x7beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  253|      1|    fill( 8, 32,  4x32,  4x16, hgtw, 0x7aeb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  254|      1|    fill( 8, 16,  4x16,  4x8,  hgtw, 0x7beb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  255|      1|    fill( 8,  8,  4x8,   4x4,  heqw, 0x7bfb);
  ------------------
  |  |  241|      1|    fill2d_16x2(w, h, BS_##w##x##h - BS_32x32, \
  |  |  242|      1|                master, wedge_codebook_16_##hvsw, \
  |  |  243|      1|                dav1d_masks.wedge_444_##w##x##h, \
  |  |  244|      1|                dav1d_masks.wedge_422_##sz_422, \
  |  |  245|      1|                dav1d_masks.wedge_420_##sz_420, signs)
  ------------------
  256|      1|#undef fill
  257|       |
  258|      1|    memset(dav1d_masks.ii_dc, 32, 32 * 32);
  259|      4|    for (int c = 0; c < 3; c++) {
  ------------------
  |  Branch (259:21): [True: 3, False: 1]
  ------------------
  260|      3|        dav1d_masks.offsets[c][BS_32x32-BS_32x32].ii[II_DC_PRED] =
  261|      3|        dav1d_masks.offsets[c][BS_32x16-BS_32x32].ii[II_DC_PRED] =
  262|      3|        dav1d_masks.offsets[c][BS_16x32-BS_32x32].ii[II_DC_PRED] =
  263|      3|        dav1d_masks.offsets[c][BS_16x16-BS_32x32].ii[II_DC_PRED] =
  264|      3|        dav1d_masks.offsets[c][BS_16x8 -BS_32x32].ii[II_DC_PRED] =
  265|      3|        dav1d_masks.offsets[c][BS_8x16 -BS_32x32].ii[II_DC_PRED] =
  266|      3|        dav1d_masks.offsets[c][BS_8x8  -BS_32x32].ii[II_DC_PRED] =
  267|      3|            MASK_OFFSET(dav1d_masks.ii_dc);
  ------------------
  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  ------------------
  268|      3|    }
  269|       |
  270|      1|#define BUILD_NONDC_II_MASKS(w, h, step) \
  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  272|       |
  273|      1|#define ASSIGN_NONDC_II_OFFSET(bs, w444, h444, w422, h422, w420, h420) \
  274|      1|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  275|      1|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  276|      1|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  277|      1|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  278|      1|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  279|      1|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  280|       |
  281|      1|    BUILD_NONDC_II_MASKS(32, 32, 1);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  282|      1|    BUILD_NONDC_II_MASKS(16, 32, 1);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  283|      1|    BUILD_NONDC_II_MASKS(16, 16, 2);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  284|      1|    BUILD_NONDC_II_MASKS( 8, 32, 1);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  285|      1|    BUILD_NONDC_II_MASKS( 8, 16, 2);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  286|      1|    BUILD_NONDC_II_MASKS( 8,  8, 4);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  287|      1|    BUILD_NONDC_II_MASKS( 4, 16, 2);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  288|      1|    BUILD_NONDC_II_MASKS( 4,  8, 4);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  289|      1|    BUILD_NONDC_II_MASKS( 4,  4, 8);
  ------------------
  |  |  271|      1|    build_nondc_ii_masks(dav1d_masks.ii_nondc_##w##x##h, w, h, step)
  ------------------
  290|      4|    for (int p = 0; p < 3; p++) {
  ------------------
  |  Branch (290:21): [True: 3, False: 1]
  ------------------
  291|      3|        ASSIGN_NONDC_II_OFFSET(BS_32x32, 32, 32, 16, 32, 16, 16);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  292|      3|        ASSIGN_NONDC_II_OFFSET(BS_32x16, 32, 32, 16, 16, 16, 16);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  293|      3|        ASSIGN_NONDC_II_OFFSET(BS_16x32, 16, 32,  8, 32,  8, 16);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  294|      3|        ASSIGN_NONDC_II_OFFSET(BS_16x16, 16, 16,  8, 16,  8,  8);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  295|      3|        ASSIGN_NONDC_II_OFFSET(BS_16x8,  16, 16,  8,  8,  8,  8);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  296|      3|        ASSIGN_NONDC_II_OFFSET(BS_8x16,   8, 16,  4, 16,  4,  8);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  297|      3|        ASSIGN_NONDC_II_OFFSET(BS_8x8,    8,  8,  4,  8,  4,  4);
  ------------------
  |  |  274|      3|    dav1d_masks.offsets[0][bs-BS_32x32].ii[p + 1] = \
  |  |  275|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w444##x##h444[p*w444*h444]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  276|      3|    dav1d_masks.offsets[1][bs-BS_32x32].ii[p + 1] = \
  |  |  277|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w422##x##h422[p*w422*h422]); \
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  |  |  278|      3|    dav1d_masks.offsets[2][bs-BS_32x32].ii[p + 1] = \
  |  |  279|      3|        MASK_OFFSET(&dav1d_masks.ii_nondc_##w420##x##h420[p*w420*h420])
  |  |  ------------------
  |  |  |  |  129|      3|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  |  |  ------------------
  ------------------
  298|      3|    }
  299|      1|}
wedge.c:insert_border:
   90|    128|{
   91|    128|    if (ctr > 4) memset(dst, 0, ctr - 4);
  ------------------
  |  Branch (91:9): [True: 128, False: 0]
  ------------------
   92|    128|    memcpy(dst + imax(ctr, 4) - 4, src + imax(4 - ctr, 0), imin(64 - ctr, 8));
   93|    128|    if (ctr < 64 - 4)
  ------------------
  |  Branch (93:9): [True: 128, False: 0]
  ------------------
   94|    128|        memset(dst + ctr + 4, 64, 64 - 4 - ctr);
   95|    128|}
wedge.c:transpose:
   97|      2|static void transpose(uint8_t *const dst, const uint8_t *const src) {
   98|    130|    for (int y = 0, y_off = 0; y < 64; y++, y_off += 64)
  ------------------
  |  Branch (98:32): [True: 128, False: 2]
  ------------------
   99|  8.32k|        for (int x = 0, x_off = 0; x < 64; x++, x_off += 64)
  ------------------
  |  Branch (99:36): [True: 8.19k, False: 128]
  ------------------
  100|  8.19k|            dst[x_off + y] = src[y_off + x];
  101|      2|}
wedge.c:hflip:
  103|      2|static void hflip(uint8_t *const dst, const uint8_t *const src) {
  104|    130|    for (int y = 0, y_off = 0; y < 64; y++, y_off += 64)
  ------------------
  |  Branch (104:32): [True: 128, False: 2]
  ------------------
  105|  8.32k|        for (int x = 0; x < 64; x++)
  ------------------
  |  Branch (105:25): [True: 8.19k, False: 128]
  ------------------
  106|  8.19k|            dst[y_off + 64 - 1 - x] = src[y_off + x];
  107|      2|}
wedge.c:fill2d_16x2:
  153|      9|{
  154|      9|    const int n_stride_444 = (w * h);
  155|      9|    const int n_stride_422 = n_stride_444 >> 1;
  156|      9|    const int n_stride_420 = n_stride_444 >> 2;
  157|      9|    const int sign_stride_422 = 16 * n_stride_422;
  158|      9|    const int sign_stride_420 = 16 * n_stride_420;
  159|       |
  160|       |    // assign pointer offsets in lookup table
  161|    153|    for (int n = 0; n < 16; n++) {
  ------------------
  |  Branch (161:21): [True: 144, False: 9]
  ------------------
  162|    144|        const int sign = signs & 1;
  163|       |
  164|    144|        copy2d(masks_444, master[cb[n].direction], sign, w, h,
  165|    144|               32 - (w * cb[n].x_offset >> 3), 32 - (h * cb[n].y_offset >> 3));
  166|       |
  167|       |        // not using !sign is intentional here, since 444 does not require
  168|       |        // any rounding since no chroma subsampling is applied.
  169|    144|        dav1d_masks.offsets[0][bs].wedge[0][n] =
  170|    144|        dav1d_masks.offsets[0][bs].wedge[1][n] = MASK_OFFSET(masks_444);
  ------------------
  |  |  129|    144|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  ------------------
  171|       |
  172|    144|        dav1d_masks.offsets[1][bs].wedge[0][n] =
  173|    144|            init_chroma(&masks_422[ sign * sign_stride_422], masks_444, 0, w, h, 0);
  174|    144|        dav1d_masks.offsets[1][bs].wedge[1][n] =
  175|    144|            init_chroma(&masks_422[!sign * sign_stride_422], masks_444, 1, w, h, 0);
  176|    144|        dav1d_masks.offsets[2][bs].wedge[0][n] =
  177|    144|            init_chroma(&masks_420[ sign * sign_stride_420], masks_444, 0, w, h, 1);
  178|    144|        dav1d_masks.offsets[2][bs].wedge[1][n] =
  179|    144|            init_chroma(&masks_420[!sign * sign_stride_420], masks_444, 1, w, h, 1);
  180|       |
  181|    144|        signs >>= 1;
  182|    144|        masks_444 += n_stride_444;
  183|    144|        masks_422 += n_stride_422;
  184|    144|        masks_420 += n_stride_420;
  185|    144|    }
  186|      9|}
wedge.c:copy2d:
  111|    144|{
  112|    144|    src += y_off * 64 + x_off;
  113|    144|    if (sign) {
  ------------------
  |  Branch (113:9): [True: 109, False: 35]
  ------------------
  114|  2.14k|        for (int y = 0; y < h; y++) {
  ------------------
  |  Branch (114:25): [True: 2.03k, False: 109]
  ------------------
  115|  40.4k|            for (int x = 0; x < w; x++)
  ------------------
  |  Branch (115:29): [True: 38.4k, False: 2.03k]
  ------------------
  116|  38.4k|                dst[x] = 64 - src[x];
  117|  2.03k|            src += 64;
  118|  2.03k|            dst += w;
  119|  2.03k|        }
  120|    109|    } else {
  121|    691|        for (int y = 0; y < h; y++) {
  ------------------
  |  Branch (121:25): [True: 656, False: 35]
  ------------------
  122|    656|            memcpy(dst, src, w);
  123|    656|            src += 64;
  124|    656|            dst += w;
  125|    656|        }
  126|     35|    }
  127|    144|}
wedge.c:init_chroma:
  134|    576|{
  135|    576|    const uint16_t offset = MASK_OFFSET(chroma);
  ------------------
  |  |  129|    576|#define MASK_OFFSET(x) ((uint16_t)(((uintptr_t)(x) - (uintptr_t)&dav1d_masks) >> 3))
  ------------------
  136|  8.64k|    for (int y = 0; y < h; y += 1 + ss_ver) {
  ------------------
  |  Branch (136:21): [True: 8.06k, False: 576]
  ------------------
  137|  83.3k|        for (int x = 0; x < w; x += 2) {
  ------------------
  |  Branch (137:25): [True: 75.2k, False: 8.06k]
  ------------------
  138|  75.2k|            int sum = luma[x] + luma[x + 1] + 1;
  139|  75.2k|            if (ss_ver) sum += luma[w + x] + luma[w + x + 1] + 1;
  ------------------
  |  Branch (139:17): [True: 25.0k, False: 50.1k]
  ------------------
  140|  75.2k|            chroma[x >> 1] = (sum - sign) >> (1 + ss_ver);
  141|  75.2k|        }
  142|  8.06k|        luma += w << ss_ver;
  143|  8.06k|        chroma += w >> 1;
  144|  8.06k|    }
  145|    576|    return offset;
  146|    576|}
wedge.c:build_nondc_ii_masks:
  190|      9|{
  191|      9|    static const uint8_t ii_weights_1d[32] = {
  192|      9|        60, 52, 45, 39, 34, 30, 26, 22, 19, 17, 15, 13, 11, 10,  8,  7,
  193|      9|         6,  6,  5,  4,  4,  3,  3,  2,  2,  2,  2,  1,  1,  1,  1,  1,
  194|      9|    };
  195|       |
  196|      9|    uint8_t *const mask_h  = &mask_v[w * h];
  197|      9|    uint8_t *const mask_sm = &mask_h[w * h];
  198|    173|    for (int y = 0, off = 0; y < h; y++, off += w) {
  ------------------
  |  Branch (198:30): [True: 164, False: 9]
  ------------------
  199|    164|        memset(&mask_v[off], ii_weights_1d[y * step], w);
  200|  2.51k|        for (int x = 0; x < w; x++) {
  ------------------
  |  Branch (200:25): [True: 2.35k, False: 164]
  ------------------
  201|  2.35k|            mask_sm[off + x] = ii_weights_1d[imin(x, y) * step];
  202|  2.35k|            mask_h[off + x] = ii_weights_1d[x * step];
  203|  2.35k|        }
  204|    164|    }
  205|      9|}

cdef_tmpl.c:cdef_dsp_init_x86:
   46|  3.46k|static ALWAYS_INLINE void cdef_dsp_init_x86(Dav1dCdefDSPContext *const c) {
   47|  3.46k|    const unsigned flags = dav1d_get_cpu_flags();
   48|       |
   49|  3.46k|#if BITDEPTH == 8
   50|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE2)) return;
  ------------------
  |  Branch (50:9): [True: 0, False: 3.46k]
  ------------------
   51|       |
   52|  3.46k|    c->fb[0] = BF(dav1d_cdef_filter_8x8, sse2);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   53|  3.46k|    c->fb[1] = BF(dav1d_cdef_filter_4x8, sse2);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   54|  3.46k|    c->fb[2] = BF(dav1d_cdef_filter_4x4, sse2);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   55|  3.46k|#endif
   56|       |
   57|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (57:9): [True: 0, False: 3.46k]
  ------------------
   58|       |
   59|  3.46k|    c->dir = BF(dav1d_cdef_dir, ssse3);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   60|  3.46k|    c->fb[0] = BF(dav1d_cdef_filter_8x8, ssse3);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   61|  3.46k|    c->fb[1] = BF(dav1d_cdef_filter_4x8, ssse3);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   62|  3.46k|    c->fb[2] = BF(dav1d_cdef_filter_4x4, ssse3);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   63|       |
   64|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE41)) return;
  ------------------
  |  Branch (64:9): [True: 0, False: 3.46k]
  ------------------
   65|       |
   66|  3.46k|    c->dir = BF(dav1d_cdef_dir, sse4);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   67|  3.46k|#if BITDEPTH == 8
   68|  3.46k|    c->fb[0] = BF(dav1d_cdef_filter_8x8, sse4);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   69|  3.46k|    c->fb[1] = BF(dav1d_cdef_filter_4x8, sse4);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   70|  3.46k|    c->fb[2] = BF(dav1d_cdef_filter_4x4, sse4);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   71|  3.46k|#endif
   72|       |
   73|  3.46k|#if ARCH_X86_64
   74|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (74:9): [True: 0, False: 3.46k]
  ------------------
   75|       |
   76|  3.46k|    c->dir = BF(dav1d_cdef_dir, avx2);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   77|  3.46k|    c->fb[0] = BF(dav1d_cdef_filter_8x8, avx2);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   78|  3.46k|    c->fb[1] = BF(dav1d_cdef_filter_4x8, avx2);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   79|  3.46k|    c->fb[2] = BF(dav1d_cdef_filter_4x4, avx2);
  ------------------
  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   80|       |
   81|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (81:9): [True: 3.46k, False: 0]
  ------------------
   82|       |
   83|      0|    c->fb[0] = BF(dav1d_cdef_filter_8x8, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   84|      0|    c->fb[1] = BF(dav1d_cdef_filter_4x8, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   85|      0|    c->fb[2] = BF(dav1d_cdef_filter_4x4, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   86|      0|#endif
   87|      0|}

dav1d_get_cpu_flags_x86:
   47|      1|COLD unsigned dav1d_get_cpu_flags_x86(void) {
   48|      1|    union {
   49|      1|        CpuidRegisters r;
   50|      1|        struct {
   51|      1|            uint32_t max_leaf;
   52|      1|            char vendor[12];
   53|      1|        };
   54|      1|    } cpu;
   55|      1|    dav1d_cpu_cpuid(&cpu.r, 0, 0);
   56|      1|    unsigned flags = dav1d_get_default_cpu_flags();
   57|       |
   58|      1|    if (cpu.max_leaf >= 1) {
  ------------------
  |  Branch (58:9): [True: 1, False: 0]
  ------------------
   59|      1|        CpuidRegisters r;
   60|      1|        dav1d_cpu_cpuid(&r, 1, 0);
   61|      1|        const unsigned family = ((r.eax >> 8) & 0x0f) + ((r.eax >> 20) & 0xff);
   62|       |
   63|      1|        if (X(r.edx, 0x06008000)) /* CMOV/SSE/SSE2 */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   64|      1|            flags |= DAV1D_X86_CPU_FLAG_SSE2;
   65|      1|            if (X(r.ecx, 0x00000201)) /* SSE3/SSSE3 */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   66|      1|                flags |= DAV1D_X86_CPU_FLAG_SSSE3;
   67|      1|                if (X(r.ecx, 0x00080000)) /* SSE4.1 */
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   68|      1|                    flags |= DAV1D_X86_CPU_FLAG_SSE41;
   69|      1|            }
   70|      1|        }
   71|      1|#if ARCH_X86_64
   72|       |        /* We only support >128-bit SIMD on x86-64. */
   73|      1|        if (X(r.ecx, 0x18000000)) /* OSXSAVE/AVX */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   74|      1|            const uint64_t xcr0 = dav1d_cpu_xgetbv(0);
   75|      1|            if (X(xcr0, 0x00000006)) /* XMM/YMM */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   76|      1|                if (cpu.max_leaf >= 7) {
  ------------------
  |  Branch (76:21): [True: 1, False: 0]
  ------------------
   77|      1|                    dav1d_cpu_cpuid(&r, 7, 0);
   78|      1|                    if (X(r.ebx, 0x00000128)) /* BMI1/BMI2/AVX2 */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 1, False: 0]
  |  |  ------------------
  ------------------
   79|      1|                        flags |= DAV1D_X86_CPU_FLAG_AVX2;
   80|      1|                        if (X(xcr0, 0x000000e0)) /* ZMM/OPMASK */ {
  ------------------
  |  |   45|      1|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 0, False: 1]
  |  |  ------------------
  ------------------
   81|      0|                            if (X(r.ebx, 0xd0230000) && X(r.ecx, 0x00005f42))
  ------------------
  |  |   45|      0|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 0, False: 0]
  |  |  ------------------
  ------------------
                                          if (X(r.ebx, 0xd0230000) && X(r.ecx, 0x00005f42))
  ------------------
  |  |   45|      0|#define X(reg, mask) (((reg) & (mask)) == (mask))
  |  |  ------------------
  |  |  |  Branch (45:22): [True: 0, False: 0]
  |  |  ------------------
  ------------------
   82|      0|                                flags |= DAV1D_X86_CPU_FLAG_AVX512ICL;
   83|      0|                        }
   84|      1|                    }
   85|      1|                }
   86|      1|            }
   87|      1|        }
   88|      1|#endif
   89|      1|        if (!memcmp(cpu.vendor, "AuthenticAMD", sizeof(cpu.vendor))) {
  ------------------
  |  Branch (89:13): [True: 1, False: 0]
  ------------------
   90|      1|            if ((flags & DAV1D_X86_CPU_FLAG_AVX2) && family <= 0x19) {
  ------------------
  |  Branch (90:17): [True: 1, False: 0]
  |  Branch (90:54): [True: 1, False: 0]
  ------------------
   91|       |                /* Excavator, Zen, Zen+, Zen 2, Zen 3, Zen 3+, Zen 4 */
   92|      1|                flags |= DAV1D_X86_CPU_FLAG_SLOW_GATHER;
   93|      1|            }
   94|      1|        }
   95|      1|    }
   96|       |
   97|      1|    return flags;
   98|      1|}

filmgrain_tmpl.c:film_grain_dsp_init_x86:
   45|  8.57k|static ALWAYS_INLINE void film_grain_dsp_init_x86(Dav1dFilmGrainDSPContext *const c) {
   46|  8.57k|    const unsigned flags = dav1d_get_cpu_flags();
   47|       |
   48|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (48:9): [True: 0, False: 8.57k]
  ------------------
   49|       |
   50|  8.57k|    c->generate_grain_y = BF(dav1d_generate_grain_y, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   51|  8.57k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_generate_grain_uv_420, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   52|  8.57k|    c->fgy_32x32xn = BF(dav1d_fgy_32x32xn, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   53|  8.57k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_fguv_32x32xn_i420, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   54|  8.57k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_generate_grain_uv_422, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   55|  8.57k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_generate_grain_uv_444, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   56|  8.57k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_fguv_32x32xn_i422, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   57|  8.57k|    c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_fguv_32x32xn_i444, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   58|       |
   59|  8.57k|#if ARCH_X86_64
   60|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (60:9): [True: 0, False: 8.57k]
  ------------------
   61|       |
   62|  8.57k|    c->generate_grain_y = BF(dav1d_generate_grain_y, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   63|  8.57k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_generate_grain_uv_420, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   64|  8.57k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_generate_grain_uv_422, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   65|  8.57k|    c->generate_grain_uv[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_generate_grain_uv_444, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   66|       |
   67|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_SLOW_GATHER)) {
  ------------------
  |  Branch (67:9): [True: 0, False: 8.57k]
  ------------------
   68|      0|        c->fgy_32x32xn = BF(dav1d_fgy_32x32xn, avx2);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   69|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_fguv_32x32xn_i420, avx2);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   70|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_fguv_32x32xn_i422, avx2);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   71|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_fguv_32x32xn_i444, avx2);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   72|      0|    }
   73|       |
   74|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (74:9): [True: 8.57k, False: 0]
  ------------------
   75|       |
   76|      0|    if (BITDEPTH == 8 || !(flags & DAV1D_X86_CPU_FLAG_SLOW_GATHER)) {
  ------------------
  |  Branch (76:9): [True: 0, Folded]
  |  Branch (76:26): [True: 0, False: 0]
  ------------------
   77|      0|        c->fgy_32x32xn = BF(dav1d_fgy_32x32xn, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   78|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I420 - 1] = BF(dav1d_fguv_32x32xn_i420, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   79|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I422 - 1] = BF(dav1d_fguv_32x32xn_i422, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   80|      0|        c->fguv_32x32xn[DAV1D_PIXEL_LAYOUT_I444 - 1] = BF(dav1d_fguv_32x32xn_i444, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   81|      0|    }
   82|      0|#endif
   83|      0|}

ipred_tmpl.c:intra_pred_dsp_init_x86:
   71|  8.57k|static ALWAYS_INLINE void intra_pred_dsp_init_x86(Dav1dIntraPredDSPContext *const c) {
   72|  8.57k|    const unsigned flags = dav1d_get_cpu_flags();
   73|       |
   74|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (74:9): [True: 0, False: 8.57k]
  ------------------
   75|       |
   76|  8.57k|    init_angular_ipred_fn(DC_PRED,       ipred_dc,       ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   77|  8.57k|    init_angular_ipred_fn(DC_128_PRED,   ipred_dc_128,   ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   78|  8.57k|    init_angular_ipred_fn(TOP_DC_PRED,   ipred_dc_top,   ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   79|  8.57k|    init_angular_ipred_fn(LEFT_DC_PRED,  ipred_dc_left,  ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   80|  8.57k|    init_angular_ipred_fn(HOR_PRED,      ipred_h,        ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   81|  8.57k|    init_angular_ipred_fn(VERT_PRED,     ipred_v,        ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   82|  8.57k|    init_angular_ipred_fn(PAETH_PRED,    ipred_paeth,    ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   83|  8.57k|    init_angular_ipred_fn(SMOOTH_PRED,   ipred_smooth,   ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   84|  8.57k|    init_angular_ipred_fn(SMOOTH_H_PRED, ipred_smooth_h, ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   85|  8.57k|    init_angular_ipred_fn(SMOOTH_V_PRED, ipred_smooth_v, ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   86|  8.57k|    init_angular_ipred_fn(Z1_PRED,       ipred_z1,       ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   87|  8.57k|    init_angular_ipred_fn(Z2_PRED,       ipred_z2,       ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   88|  8.57k|    init_angular_ipred_fn(Z3_PRED,       ipred_z3,       ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   89|  8.57k|    init_angular_ipred_fn(FILTER_PRED,   ipred_filter,   ssse3);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   90|       |
   91|  8.57k|    init_cfl_pred_fn(DC_PRED,      ipred_cfl,      ssse3);
  ------------------
  |  |   41|  8.57k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   92|  8.57k|    init_cfl_pred_fn(DC_128_PRED,  ipred_cfl_128,  ssse3);
  ------------------
  |  |   41|  8.57k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   93|  8.57k|    init_cfl_pred_fn(TOP_DC_PRED,  ipred_cfl_top,  ssse3);
  ------------------
  |  |   41|  8.57k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   94|  8.57k|    init_cfl_pred_fn(LEFT_DC_PRED, ipred_cfl_left, ssse3);
  ------------------
  |  |   41|  8.57k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   95|       |
   96|  8.57k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I420 - 1, ipred_cfl_ac_420, ssse3);
  ------------------
  |  |   43|  8.57k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   97|  8.57k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I422 - 1, ipred_cfl_ac_422, ssse3);
  ------------------
  |  |   43|  8.57k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   98|  8.57k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I444 - 1, ipred_cfl_ac_444, ssse3);
  ------------------
  |  |   43|  8.57k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   99|       |
  100|  8.57k|    c->pal_pred = BF(dav1d_pal_pred, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  101|       |
  102|  8.57k|#if ARCH_X86_64
  103|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (103:9): [True: 0, False: 8.57k]
  ------------------
  104|       |
  105|  8.57k|    init_angular_ipred_fn(DC_PRED,       ipred_dc,       avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  106|  8.57k|    init_angular_ipred_fn(DC_128_PRED,   ipred_dc_128,   avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  107|  8.57k|    init_angular_ipred_fn(TOP_DC_PRED,   ipred_dc_top,   avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  108|  8.57k|    init_angular_ipred_fn(LEFT_DC_PRED,  ipred_dc_left,  avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  109|  8.57k|    init_angular_ipred_fn(HOR_PRED,      ipred_h,        avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  110|  8.57k|    init_angular_ipred_fn(VERT_PRED,     ipred_v,        avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  111|  8.57k|    init_angular_ipred_fn(PAETH_PRED,    ipred_paeth,    avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  112|  8.57k|    init_angular_ipred_fn(SMOOTH_PRED,   ipred_smooth,   avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  113|  8.57k|    init_angular_ipred_fn(SMOOTH_H_PRED, ipred_smooth_h, avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  114|  8.57k|    init_angular_ipred_fn(SMOOTH_V_PRED, ipred_smooth_v, avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  115|  8.57k|    init_angular_ipred_fn(Z1_PRED,       ipred_z1,       avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  116|  8.57k|    init_angular_ipred_fn(Z2_PRED,       ipred_z2,       avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  117|  8.57k|    init_angular_ipred_fn(Z3_PRED,       ipred_z3,       avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  118|  8.57k|    init_angular_ipred_fn(FILTER_PRED,   ipred_filter,   avx2);
  ------------------
  |  |   39|  8.57k|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  119|       |
  120|  8.57k|    init_cfl_pred_fn(DC_PRED,      ipred_cfl,      avx2);
  ------------------
  |  |   41|  8.57k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  121|  8.57k|    init_cfl_pred_fn(DC_128_PRED,  ipred_cfl_128,  avx2);
  ------------------
  |  |   41|  8.57k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  122|  8.57k|    init_cfl_pred_fn(TOP_DC_PRED,  ipred_cfl_top,  avx2);
  ------------------
  |  |   41|  8.57k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  123|  8.57k|    init_cfl_pred_fn(LEFT_DC_PRED, ipred_cfl_left, avx2);
  ------------------
  |  |   41|  8.57k|    init_fn(cfl_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  124|       |
  125|  8.57k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I420 - 1, ipred_cfl_ac_420, avx2);
  ------------------
  |  |   43|  8.57k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  126|  8.57k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I422 - 1, ipred_cfl_ac_422, avx2);
  ------------------
  |  |   43|  8.57k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  127|  8.57k|    init_cfl_ac_fn(DAV1D_PIXEL_LAYOUT_I444 - 1, ipred_cfl_ac_444, avx2);
  ------------------
  |  |   43|  8.57k|    init_fn(cfl_ac, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  128|       |
  129|  8.57k|    c->pal_pred = BF(dav1d_pal_pred, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  130|       |
  131|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (131:9): [True: 8.57k, False: 0]
  ------------------
  132|       |
  133|      0|#if BITDEPTH == 8
  134|      0|    init_angular_ipred_fn(DC_PRED,       ipred_dc,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|  8.57k|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  135|      0|    init_angular_ipred_fn(DC_128_PRED,   ipred_dc_128,   avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  136|      0|    init_angular_ipred_fn(TOP_DC_PRED,   ipred_dc_top,   avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  137|      0|    init_angular_ipred_fn(LEFT_DC_PRED,  ipred_dc_left,  avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  138|      0|    init_angular_ipred_fn(HOR_PRED,      ipred_h,        avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  139|      0|    init_angular_ipred_fn(VERT_PRED,     ipred_v,        avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  140|      0|    init_angular_ipred_fn(Z2_PRED,       ipred_z2,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  141|      0|#endif
  142|      0|    init_angular_ipred_fn(PAETH_PRED,    ipred_paeth,    avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  143|      0|    init_angular_ipred_fn(SMOOTH_PRED,   ipred_smooth,   avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  144|      0|    init_angular_ipred_fn(SMOOTH_H_PRED, ipred_smooth_h, avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  145|      0|    init_angular_ipred_fn(SMOOTH_V_PRED, ipred_smooth_v, avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  146|      0|    init_angular_ipred_fn(Z1_PRED,       ipred_z1,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  147|      0|    init_angular_ipred_fn(Z2_PRED,       ipred_z2,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  148|      0|    init_angular_ipred_fn(Z3_PRED,       ipred_z3,       avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  149|      0|    init_angular_ipred_fn(FILTER_PRED,   ipred_filter,   avx512icl);
  ------------------
  |  |   39|      0|    init_fn(intra_pred, type, name, suffix)
  |  |  ------------------
  |  |  |  |   36|      0|    c->type0[type1] = BF(dav1d_##name, suffix)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  150|       |
  151|      0|    c->pal_pred = BF(dav1d_pal_pred, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  152|      0|#endif
  153|      0|}

itx_tmpl.c:itx_dsp_init_x86:
  112|  3.46k|{
  113|  3.46k|#define assign_itx_bpc_fn(pfx, w, h, type, type_enum, bpc, ext) \
  114|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  115|  3.46k|        BF_BPC(dav1d_inv_txfm_add_##type##_##w##x##h, bpc, ext)
  116|       |
  117|  3.46k|#define assign_itx1_bpc_fn(pfx, w, h, bpc, ext) \
  118|  3.46k|    assign_itx_bpc_fn(pfx, w, h, dct_dct,           DCT_DCT,           bpc, ext)
  119|       |
  120|  3.46k|#define assign_itx2_bpc_fn(pfx, w, h, bpc, ext) \
  121|  3.46k|    assign_itx1_bpc_fn(pfx, w, h, bpc, ext); \
  122|  3.46k|    assign_itx_bpc_fn(pfx, w, h, identity_identity, IDTX,              bpc, ext)
  123|       |
  124|  3.46k|#define assign_itx12_bpc_fn(pfx, w, h, bpc, ext) \
  125|  3.46k|    assign_itx2_bpc_fn(pfx, w, h, bpc, ext); \
  126|  3.46k|    assign_itx_bpc_fn(pfx, w, h, dct_adst,          ADST_DCT,          bpc, ext); \
  127|  3.46k|    assign_itx_bpc_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      bpc, ext); \
  128|  3.46k|    assign_itx_bpc_fn(pfx, w, h, dct_identity,      H_DCT,             bpc, ext); \
  129|  3.46k|    assign_itx_bpc_fn(pfx, w, h, adst_dct,          DCT_ADST,          bpc, ext); \
  130|  3.46k|    assign_itx_bpc_fn(pfx, w, h, adst_adst,         ADST_ADST,         bpc, ext); \
  131|  3.46k|    assign_itx_bpc_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     bpc, ext); \
  132|  3.46k|    assign_itx_bpc_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      bpc, ext); \
  133|  3.46k|    assign_itx_bpc_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     bpc, ext); \
  134|  3.46k|    assign_itx_bpc_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, bpc, ext); \
  135|  3.46k|    assign_itx_bpc_fn(pfx, w, h, identity_dct,      V_DCT,             bpc, ext)
  136|       |
  137|  3.46k|#define assign_itx16_bpc_fn(pfx, w, h, bpc, ext) \
  138|  3.46k|    assign_itx12_bpc_fn(pfx, w, h, bpc, ext); \
  139|  3.46k|    assign_itx_bpc_fn(pfx, w, h, adst_identity,     H_ADST,            bpc, ext); \
  140|  3.46k|    assign_itx_bpc_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        bpc, ext); \
  141|  3.46k|    assign_itx_bpc_fn(pfx, w, h, identity_adst,     V_ADST,            bpc, ext); \
  142|  3.46k|    assign_itx_bpc_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        bpc, ext)
  143|       |
  144|  3.46k|    const unsigned flags = dav1d_get_cpu_flags();
  145|       |
  146|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE2)) return;
  ------------------
  |  Branch (146:9): [True: 0, False: 3.46k]
  ------------------
  147|       |
  148|  3.46k|    assign_itx_fn(, 4, 4, wht_wht, WHT_WHT, sse2);
  ------------------
  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  ------------------
  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  149|       |
  150|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (150:9): [True: 0, False: 3.46k]
  ------------------
  151|       |
  152|  3.46k|#if BITDEPTH == 8
  153|  3.46k|    assign_itx16_fn(,   4,  4, ssse3);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  154|  3.46k|    assign_itx16_fn(R,  4,  8, ssse3);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  155|  3.46k|    assign_itx16_fn(R,  8,  4, ssse3);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  156|  3.46k|    assign_itx16_fn(,   8,  8, ssse3);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  157|  3.46k|    assign_itx16_fn(R,  4, 16, ssse3);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  158|  3.46k|    assign_itx16_fn(R, 16,  4, ssse3);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  159|  3.46k|    assign_itx16_fn(R,  8, 16, ssse3);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  160|  3.46k|    assign_itx16_fn(R, 16,  8, ssse3);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  161|  3.46k|    assign_itx12_fn(,  16, 16, ssse3);
  ------------------
  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  162|  3.46k|    assign_itx2_fn (R,  8, 32, ssse3);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  163|  3.46k|    assign_itx2_fn (R, 32,  8, ssse3);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  164|  3.46k|    assign_itx2_fn (R, 16, 32, ssse3);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  165|  3.46k|    assign_itx2_fn (R, 32, 16, ssse3);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  166|  3.46k|    assign_itx2_fn (,  32, 32, ssse3);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  167|  3.46k|    assign_itx1_fn (R, 16, 64, ssse3);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  168|  3.46k|    assign_itx1_fn (R, 32, 64, ssse3);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  169|  3.46k|    assign_itx1_fn (R, 64, 16, ssse3);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  170|  3.46k|    assign_itx1_fn (R, 64, 32, ssse3);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  171|  3.46k|    assign_itx1_fn ( , 64, 64, ssse3);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  172|  3.46k|    *all_simd = 1;
  173|  3.46k|#endif
  174|       |
  175|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE41)) return;
  ------------------
  |  Branch (175:9): [True: 0, False: 3.46k]
  ------------------
  176|       |
  177|       |#if BITDEPTH == 16
  178|       |    if (bpc == 10) {
  179|       |        assign_itx16_fn(,   4,  4, sse4);
  180|       |        assign_itx16_fn(R,  4,  8, sse4);
  181|       |        assign_itx16_fn(R,  4, 16, sse4);
  182|       |        assign_itx16_fn(R,  8,  4, sse4);
  183|       |        assign_itx16_fn(,   8,  8, sse4);
  184|       |        assign_itx16_fn(R,  8, 16, sse4);
  185|       |        assign_itx16_fn(R, 16,  4, sse4);
  186|       |        assign_itx16_fn(R, 16,  8, sse4);
  187|       |        assign_itx12_fn(,  16, 16, sse4);
  188|       |        assign_itx2_fn (R,  8, 32, sse4);
  189|       |        assign_itx2_fn (R, 32,  8, sse4);
  190|       |        assign_itx2_fn (R, 16, 32, sse4);
  191|       |        assign_itx2_fn (R, 32, 16, sse4);
  192|       |        assign_itx2_fn (,  32, 32, sse4);
  193|       |        assign_itx1_fn (R, 16, 64, sse4);
  194|       |        assign_itx1_fn (R, 32, 64, sse4);
  195|       |        assign_itx1_fn (R, 64, 16, sse4);
  196|       |        assign_itx1_fn (R, 64, 32, sse4);
  197|       |        assign_itx1_fn (,  64, 64, sse4);
  198|       |        *all_simd = 1;
  199|       |    }
  200|       |#endif
  201|       |
  202|  3.46k|#if ARCH_X86_64
  203|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (203:9): [True: 0, False: 3.46k]
  ------------------
  204|       |
  205|  3.46k|    assign_itx_fn(, 4, 4, wht_wht, WHT_WHT, avx2);
  ------------------
  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  ------------------
  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  206|       |
  207|  3.46k|#if BITDEPTH == 8
  208|  3.46k|    assign_itx16_fn( ,  4,  4, avx2);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  209|  3.46k|    assign_itx16_fn(R,  4,  8, avx2);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  210|  3.46k|    assign_itx16_fn(R,  4, 16, avx2);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  211|  3.46k|    assign_itx16_fn(R,  8,  4, avx2);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  212|  3.46k|    assign_itx16_fn( ,  8,  8, avx2);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  213|  3.46k|    assign_itx16_fn(R,  8, 16, avx2);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  214|  3.46k|    assign_itx2_fn (R,  8, 32, avx2);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  215|  3.46k|    assign_itx16_fn(R, 16,  4, avx2);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  216|  3.46k|    assign_itx16_fn(R, 16,  8, avx2);
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|  3.46k|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|  3.46k|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  217|  3.46k|    assign_itx12_fn( , 16, 16, avx2);
  ------------------
  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   89|  3.46k|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   90|  3.46k|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   91|  3.46k|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   92|  3.46k|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   93|  3.46k|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   94|  3.46k|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   95|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   96|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   97|  3.46k|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  218|  3.46k|    assign_itx2_fn (R, 16, 32, avx2);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  219|  3.46k|    assign_itx1_fn (R, 16, 64, avx2);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  220|  3.46k|    assign_itx2_fn (R, 32,  8, avx2);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  221|  3.46k|    assign_itx2_fn (R, 32, 16, avx2);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  222|  3.46k|    assign_itx2_fn ( , 32, 32, avx2);
  ------------------
  |  |   84|  3.46k|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  223|  3.46k|    assign_itx1_fn (R, 32, 64, avx2);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  224|  3.46k|    assign_itx1_fn (R, 64, 16, avx2);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  225|  3.46k|    assign_itx1_fn (R, 64, 32, avx2);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  226|  3.46k|    assign_itx1_fn ( , 64, 64, avx2);
  ------------------
  |  |   81|  3.46k|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  227|       |#else
  228|       |    if (bpc == 10) {
  229|       |        assign_itx16_bpc_fn( ,  4,  4, 10, avx2);
  230|       |        assign_itx16_bpc_fn(R,  4,  8, 10, avx2);
  231|       |        assign_itx16_bpc_fn(R,  4, 16, 10, avx2);
  232|       |        assign_itx16_bpc_fn(R,  8,  4, 10, avx2);
  233|       |        assign_itx16_bpc_fn( ,  8,  8, 10, avx2);
  234|       |        assign_itx16_bpc_fn(R,  8, 16, 10, avx2);
  235|       |        assign_itx2_bpc_fn (R,  8, 32, 10, avx2);
  236|       |        assign_itx16_bpc_fn(R, 16,  4, 10, avx2);
  237|       |        assign_itx16_bpc_fn(R, 16,  8, 10, avx2);
  238|       |        assign_itx12_bpc_fn( , 16, 16, 10, avx2);
  239|       |        assign_itx2_bpc_fn (R, 16, 32, 10, avx2);
  240|       |        assign_itx1_bpc_fn (R, 16, 64, 10, avx2);
  241|       |        assign_itx2_bpc_fn (R, 32,  8, 10, avx2);
  242|       |        assign_itx2_bpc_fn (R, 32, 16, 10, avx2);
  243|       |        assign_itx2_bpc_fn ( , 32, 32, 10, avx2);
  244|       |        assign_itx1_bpc_fn (R, 32, 64, 10, avx2);
  245|       |        assign_itx1_bpc_fn (R, 64, 16, 10, avx2);
  246|       |        assign_itx1_bpc_fn (R, 64, 32, 10, avx2);
  247|       |        assign_itx1_bpc_fn ( , 64, 64, 10, avx2);
  248|       |    } else {
  249|       |        assign_itx16_bpc_fn( ,  4,  4, 12, avx2);
  250|       |        assign_itx16_bpc_fn(R,  4,  8, 12, avx2);
  251|       |        assign_itx16_bpc_fn(R,  4, 16, 12, avx2);
  252|       |        assign_itx16_bpc_fn(R,  8,  4, 12, avx2);
  253|       |        assign_itx16_bpc_fn( ,  8,  8, 12, avx2);
  254|       |        assign_itx16_bpc_fn(R,  8, 16, 12, avx2);
  255|       |        assign_itx2_bpc_fn (R,  8, 32, 12, avx2);
  256|       |        assign_itx16_bpc_fn(R, 16,  4, 12, avx2);
  257|       |        assign_itx16_bpc_fn(R, 16,  8, 12, avx2);
  258|       |        assign_itx12_bpc_fn( , 16, 16, 12, avx2);
  259|       |        assign_itx2_bpc_fn (R, 32,  8, 12, avx2);
  260|       |        assign_itx_bpc_fn(R, 16, 32, identity_identity, IDTX, 12, avx2);
  261|       |        assign_itx_bpc_fn(R, 32, 16, identity_identity, IDTX, 12, avx2);
  262|       |        assign_itx_bpc_fn( , 32, 32, identity_identity, IDTX, 12, avx2);
  263|       |    }
  264|       |#endif
  265|       |
  266|  3.46k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (266:9): [True: 3.46k, False: 0]
  ------------------
  267|       |
  268|      0|#if BITDEPTH == 8
  269|  3.46k|    assign_itx16_fn( ,  4,  4, avx512icl); // no wht
  ------------------
  |  |  101|  3.46k|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|  3.46k|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|  3.46k|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|  3.46k|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|  3.46k|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|  3.46k|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|  3.46k|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|  3.46k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  270|      0|    assign_itx16_fn(R,  4,  8, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  271|      0|    assign_itx16_fn(R,  4, 16, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  272|      0|    assign_itx16_fn(R,  8,  4, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  273|      0|    assign_itx16_fn( ,  8,  8, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  274|      0|    assign_itx16_fn(R,  8, 16, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  275|      0|    assign_itx2_fn (R,  8, 32, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  276|      0|    assign_itx16_fn(R, 16,  4, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  277|      0|    assign_itx16_fn(R, 16,  8, avx512icl);
  ------------------
  |  |  101|      0|    assign_itx12_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  102|      0|    assign_itx_fn(pfx, w, h, adst_identity,     H_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  103|      0|    assign_itx_fn(pfx, w, h, flipadst_identity, H_FLIPADST,        ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  104|      0|    assign_itx_fn(pfx, w, h, identity_adst,     V_ADST,            ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  105|      0|    assign_itx_fn(pfx, w, h, identity_flipadst, V_FLIPADST,        ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  278|      0|    assign_itx12_fn( , 16, 16, avx512icl);
  ------------------
  |  |   88|      0|    assign_itx2_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   89|      0|    assign_itx_fn(pfx, w, h, dct_adst,          ADST_DCT,          ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   90|      0|    assign_itx_fn(pfx, w, h, dct_flipadst,      FLIPADST_DCT,      ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   91|      0|    assign_itx_fn(pfx, w, h, dct_identity,      H_DCT,             ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   92|      0|    assign_itx_fn(pfx, w, h, adst_dct,          DCT_ADST,          ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   93|      0|    assign_itx_fn(pfx, w, h, adst_adst,         ADST_ADST,         ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   94|      0|    assign_itx_fn(pfx, w, h, adst_flipadst,     FLIPADST_ADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   95|      0|    assign_itx_fn(pfx, w, h, flipadst_dct,      DCT_FLIPADST,      ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   96|      0|    assign_itx_fn(pfx, w, h, flipadst_adst,     ADST_FLIPADST,     ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   97|      0|    assign_itx_fn(pfx, w, h, flipadst_flipadst, FLIPADST_FLIPADST, ext); \
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   98|      0|    assign_itx_fn(pfx, w, h, identity_dct,      V_DCT,             ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  279|      0|    assign_itx2_fn (R, 16, 32, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  280|      0|    assign_itx1_fn (R, 16, 64, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  281|      0|    assign_itx2_fn (R, 32,  8, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  282|      0|    assign_itx2_fn (R, 32, 16, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  283|      0|    assign_itx2_fn ( , 32, 32, avx512icl);
  ------------------
  |  |   84|      0|    assign_itx1_fn(pfx, w, h, ext); \
  |  |  ------------------
  |  |  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |   85|      0|    assign_itx_fn(pfx, w, h, identity_identity, IDTX,              ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  284|      0|    assign_itx1_fn (R, 32, 64, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  285|      0|    assign_itx1_fn (R, 64, 16, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  286|      0|    assign_itx1_fn (R, 64, 32, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  287|      0|    assign_itx1_fn ( , 64, 64, avx512icl);
  ------------------
  |  |   81|      0|    assign_itx_fn(pfx, w, h, dct_dct,           DCT_DCT,           ext)
  |  |  ------------------
  |  |  |  |   77|      0|    c->itxfm_add[pfx##TX_##w##X##h][type_enum] = \
  |  |  |  |   78|      0|        BF(dav1d_inv_txfm_add_##type##_##w##x##h, ext)
  |  |  |  |  ------------------
  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  288|       |#else
  289|       |    if (bpc == 10) {
  290|       |        assign_itx16_bpc_fn( ,  8,  8, 10, avx512icl);
  291|       |        assign_itx16_bpc_fn(R,  8, 16, 10, avx512icl);
  292|       |        assign_itx2_bpc_fn (R,  8, 32, 10, avx512icl);
  293|       |        assign_itx16_bpc_fn(R, 16,  8, 10, avx512icl);
  294|       |        assign_itx12_bpc_fn( , 16, 16, 10, avx512icl);
  295|       |        assign_itx2_bpc_fn (R, 16, 32, 10, avx512icl);
  296|       |        assign_itx2_bpc_fn (R, 32,  8, 10, avx512icl);
  297|       |        assign_itx2_bpc_fn (R, 32, 16, 10, avx512icl);
  298|       |        assign_itx2_bpc_fn ( , 32, 32, 10, avx512icl);
  299|       |        assign_itx1_bpc_fn (R, 16, 64, 10, avx512icl);
  300|       |        assign_itx1_bpc_fn (R, 32, 64, 10, avx512icl);
  301|       |        assign_itx1_bpc_fn (R, 64, 16, 10, avx512icl);
  302|       |        assign_itx1_bpc_fn (R, 64, 32, 10, avx512icl);
  303|       |        assign_itx1_bpc_fn ( , 64, 64, 10, avx512icl);
  304|       |    }
  305|       |#endif
  306|      0|#endif
  307|      0|}

loopfilter_tmpl.c:loop_filter_dsp_init_x86:
   41|  8.57k|static ALWAYS_INLINE void loop_filter_dsp_init_x86(Dav1dLoopFilterDSPContext *const c) {
   42|  8.57k|    const unsigned flags = dav1d_get_cpu_flags();
   43|       |
   44|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (44:9): [True: 0, False: 8.57k]
  ------------------
   45|       |
   46|  8.57k|    c->loop_filter_sb[0][0] = BF(dav1d_lpf_h_sb_y, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   47|  8.57k|    c->loop_filter_sb[0][1] = BF(dav1d_lpf_v_sb_y, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   48|  8.57k|    c->loop_filter_sb[1][0] = BF(dav1d_lpf_h_sb_uv, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   49|  8.57k|    c->loop_filter_sb[1][1] = BF(dav1d_lpf_v_sb_uv, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   50|       |
   51|  8.57k|#if ARCH_X86_64
   52|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (52:9): [True: 0, False: 8.57k]
  ------------------
   53|       |
   54|  8.57k|    c->loop_filter_sb[0][0] = BF(dav1d_lpf_h_sb_y, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   55|  8.57k|    c->loop_filter_sb[0][1] = BF(dav1d_lpf_v_sb_y, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   56|  8.57k|    c->loop_filter_sb[1][0] = BF(dav1d_lpf_h_sb_uv, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   57|  8.57k|    c->loop_filter_sb[1][1] = BF(dav1d_lpf_v_sb_uv, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   58|       |
   59|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (59:9): [True: 8.57k, False: 0]
  ------------------
   60|       |
   61|      0|    c->loop_filter_sb[0][1] = BF(dav1d_lpf_v_sb_y, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   62|      0|    c->loop_filter_sb[1][1] = BF(dav1d_lpf_v_sb_uv, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   63|       |
   64|      0|    if (!(flags & DAV1D_X86_CPU_FLAG_SLOW_GATHER)) {
  ------------------
  |  Branch (64:9): [True: 0, False: 0]
  ------------------
   65|      0|        c->loop_filter_sb[0][0] = BF(dav1d_lpf_h_sb_y, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   66|      0|        c->loop_filter_sb[1][0] = BF(dav1d_lpf_h_sb_uv, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   67|      0|    }
   68|      0|#endif
   69|      0|}

looprestoration_tmpl.c:loop_restoration_dsp_init_x86:
   50|  8.57k|static ALWAYS_INLINE void loop_restoration_dsp_init_x86(Dav1dLoopRestorationDSPContext *const c, const int bpc) {
   51|  8.57k|    const unsigned flags = dav1d_get_cpu_flags();
   52|       |
   53|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE2)) return;
  ------------------
  |  Branch (53:9): [True: 0, False: 8.57k]
  ------------------
   54|  8.57k|#if BITDEPTH == 8
   55|  8.57k|    c->wiener[0] = BF(dav1d_wiener_filter7, sse2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   56|  8.57k|    c->wiener[1] = BF(dav1d_wiener_filter5, sse2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   57|  8.57k|#endif
   58|       |
   59|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (59:9): [True: 0, False: 8.57k]
  ------------------
   60|  8.57k|    c->wiener[0] = BF(dav1d_wiener_filter7, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   61|  8.57k|    c->wiener[1] = BF(dav1d_wiener_filter5, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   62|  8.57k|    if (BITDEPTH == 8 || bpc == 10) {
  ------------------
  |  Branch (62:9): [True: 3.46k, Folded]
  |  Branch (62:26): [True: 2.11k, False: 2.99k]
  ------------------
   63|  5.58k|        c->sgr[0] = BF(dav1d_sgr_filter_5x5, ssse3);
  ------------------
  |  |   52|  5.58k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   64|  5.58k|        c->sgr[1] = BF(dav1d_sgr_filter_3x3, ssse3);
  ------------------
  |  |   52|  5.58k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   65|  5.58k|        c->sgr[2] = BF(dav1d_sgr_filter_mix, ssse3);
  ------------------
  |  |   52|  5.58k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   66|  5.58k|    }
   67|       |
   68|  8.57k|#if ARCH_X86_64
   69|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (69:9): [True: 0, False: 8.57k]
  ------------------
   70|       |
   71|  8.57k|    c->wiener[0] = BF(dav1d_wiener_filter7, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   72|  8.57k|    c->wiener[1] = BF(dav1d_wiener_filter5, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   73|  8.57k|    if (BITDEPTH == 8 || bpc == 10) {
  ------------------
  |  Branch (73:9): [True: 3.46k, Folded]
  |  Branch (73:26): [True: 2.11k, False: 2.99k]
  ------------------
   74|  5.58k|        c->sgr[0] = BF(dav1d_sgr_filter_5x5, avx2);
  ------------------
  |  |   52|  5.58k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   75|  5.58k|        c->sgr[1] = BF(dav1d_sgr_filter_3x3, avx2);
  ------------------
  |  |   52|  5.58k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   76|  5.58k|        c->sgr[2] = BF(dav1d_sgr_filter_mix, avx2);
  ------------------
  |  |   52|  5.58k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   77|  5.58k|    }
   78|       |
   79|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (79:9): [True: 8.57k, False: 0]
  ------------------
   80|       |
   81|      0|    c->wiener[0] = BF(dav1d_wiener_filter7, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   82|      0|#if BITDEPTH == 8
   83|       |    /* With VNNI we don't need a 5-tap version. */
   84|      0|    c->wiener[1] = c->wiener[0];
   85|       |#else
   86|       |    c->wiener[1] = BF(dav1d_wiener_filter5, avx512icl);
   87|       |#endif
   88|      0|    if (BITDEPTH == 8 || bpc == 10) {
  ------------------
  |  Branch (88:9): [True: 0, Folded]
  |  Branch (88:26): [True: 0, False: 0]
  ------------------
   89|      0|        c->sgr[0] = BF(dav1d_sgr_filter_5x5, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   90|      0|        c->sgr[1] = BF(dav1d_sgr_filter_3x3, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   91|      0|        c->sgr[2] = BF(dav1d_sgr_filter_mix, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
   92|      0|    }
   93|      0|#endif
   94|      0|}

mc_tmpl.c:mc_dsp_init_x86:
   92|  8.57k|static ALWAYS_INLINE void mc_dsp_init_x86(Dav1dMCDSPContext *const c) {
   93|  8.57k|    const unsigned flags = dav1d_get_cpu_flags();
   94|       |
   95|  8.57k|    if(!(flags & DAV1D_X86_CPU_FLAG_SSSE3))
  ------------------
  |  Branch (95:8): [True: 0, False: 8.57k]
  ------------------
   96|      0|        return;
   97|       |
   98|  8.57k|    init_8tap_fns(ssse3);
  ------------------
  |  |  143|  8.57k|    init_8tap_gen(mc,  opt); \
  |  |  ------------------
  |  |  |  |  132|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  144|  8.57k|    init_8tap_gen(mct, opt)
  |  |  ------------------
  |  |  |  |  132|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
   99|       |
  100|  8.57k|    init_mc_fn(FILTER_2D_BILINEAR,             bilin,               ssse3);
  ------------------
  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  101|  8.57k|    init_mct_fn(FILTER_2D_BILINEAR,            bilin,               ssse3);
  ------------------
  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  102|       |
  103|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR,        8tap_scaled_regular,        ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  104|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_scaled_regular_smooth, ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  105|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_scaled_regular_sharp,  ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  106|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_scaled_smooth_regular, ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  107|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH,         8tap_scaled_smooth,         ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  108|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_scaled_smooth_sharp,   ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  109|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_scaled_sharp_regular,  ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  110|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_scaled_sharp_smooth,   ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  111|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP,          8tap_scaled_sharp,          ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  112|  8.57k|    init_mc_scaled_fn(FILTER_2D_BILINEAR,            bilin_scaled,               ssse3);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  113|       |
  114|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR,        8tap_scaled_regular,        ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  115|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_scaled_regular_smooth, ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  116|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_scaled_regular_sharp,  ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  117|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_scaled_smooth_regular, ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  118|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH,         8tap_scaled_smooth,         ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  119|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_scaled_smooth_sharp,   ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  120|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_scaled_sharp_regular,  ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  121|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_scaled_sharp_smooth,   ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  122|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP,          8tap_scaled_sharp,          ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  123|  8.57k|    init_mct_scaled_fn(FILTER_2D_BILINEAR,            bilin_scaled,               ssse3);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  124|       |
  125|  8.57k|    c->avg = BF(dav1d_avg, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  126|  8.57k|    c->w_avg = BF(dav1d_w_avg, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  127|  8.57k|    c->mask = BF(dav1d_mask, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  128|  8.57k|    c->w_mask[0] = BF(dav1d_w_mask_444, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  129|  8.57k|    c->w_mask[1] = BF(dav1d_w_mask_422, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  130|  8.57k|    c->w_mask[2] = BF(dav1d_w_mask_420, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  131|  8.57k|    c->blend = BF(dav1d_blend, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  132|  8.57k|    c->blend_v = BF(dav1d_blend_v, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  133|  8.57k|    c->blend_h = BF(dav1d_blend_h, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  134|  8.57k|    c->warp8x8  = BF(dav1d_warp_affine_8x8, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  135|  8.57k|    c->warp8x8t = BF(dav1d_warp_affine_8x8t, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  136|  8.57k|    c->emu_edge = BF(dav1d_emu_edge, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  137|  8.57k|    c->resize = BF(dav1d_resize, ssse3);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  138|       |
  139|  8.57k|    if(!(flags & DAV1D_X86_CPU_FLAG_SSE41))
  ------------------
  |  Branch (139:8): [True: 0, False: 8.57k]
  ------------------
  140|      0|        return;
  141|       |
  142|  8.57k|#if BITDEPTH == 8
  143|  8.57k|    c->warp8x8  = BF(dav1d_warp_affine_8x8, sse4);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  144|  8.57k|    c->warp8x8t = BF(dav1d_warp_affine_8x8t, sse4);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  145|  8.57k|#endif
  146|       |
  147|  8.57k|#if ARCH_X86_64
  148|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2))
  ------------------
  |  Branch (148:9): [True: 0, False: 8.57k]
  ------------------
  149|      0|        return;
  150|       |
  151|  8.57k|    init_8tap_fns(avx2);
  ------------------
  |  |  143|  8.57k|    init_8tap_gen(mc,  opt); \
  |  |  ------------------
  |  |  |  |  132|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  144|  8.57k|    init_8tap_gen(mct, opt)
  |  |  ------------------
  |  |  |  |  132|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  152|       |
  153|  8.57k|    init_mc_fn(FILTER_2D_BILINEAR,            bilin,               avx2);
  ------------------
  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  154|  8.57k|    init_mct_fn(FILTER_2D_BILINEAR,           bilin,               avx2);
  ------------------
  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  155|       |
  156|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR,        8tap_scaled_regular,        avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  157|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_scaled_regular_smooth, avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  158|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_scaled_regular_sharp,  avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  159|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_scaled_smooth_regular, avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  160|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH,         8tap_scaled_smooth,         avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  161|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_scaled_smooth_sharp,   avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  162|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_scaled_sharp_regular,  avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  163|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_scaled_sharp_smooth,   avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  164|  8.57k|    init_mc_scaled_fn(FILTER_2D_8TAP_SHARP,          8tap_scaled_sharp,          avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  165|  8.57k|    init_mc_scaled_fn(FILTER_2D_BILINEAR,            bilin_scaled,               avx2);
  ------------------
  |  |   40|  8.57k|    c->mc_scaled[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  166|       |
  167|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR,        8tap_scaled_regular,        avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  168|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_scaled_regular_smooth, avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  169|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_scaled_regular_sharp,  avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  170|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_scaled_smooth_regular, avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  171|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH,         8tap_scaled_smooth,         avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  172|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_scaled_smooth_sharp,   avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  173|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_scaled_sharp_regular,  avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  174|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_scaled_sharp_smooth,   avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  175|  8.57k|    init_mct_scaled_fn(FILTER_2D_8TAP_SHARP,          8tap_scaled_sharp,          avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  176|  8.57k|    init_mct_scaled_fn(FILTER_2D_BILINEAR,            bilin_scaled,               avx2);
  ------------------
  |  |   42|  8.57k|    c->mct_scaled[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  177|       |
  178|  8.57k|    c->avg = BF(dav1d_avg, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  179|  8.57k|    c->w_avg = BF(dav1d_w_avg, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  180|  8.57k|    c->mask = BF(dav1d_mask, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  181|  8.57k|    c->w_mask[0] = BF(dav1d_w_mask_444, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  182|  8.57k|    c->w_mask[1] = BF(dav1d_w_mask_422, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  183|  8.57k|    c->w_mask[2] = BF(dav1d_w_mask_420, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  184|  8.57k|    c->blend = BF(dav1d_blend, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  185|  8.57k|    c->blend_v = BF(dav1d_blend_v, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  186|  8.57k|    c->blend_h = BF(dav1d_blend_h, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  187|  8.57k|    c->warp8x8  = BF(dav1d_warp_affine_8x8, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  188|  8.57k|    c->warp8x8t = BF(dav1d_warp_affine_8x8t, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  189|  8.57k|    c->emu_edge = BF(dav1d_emu_edge, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  190|  8.57k|    c->resize = BF(dav1d_resize, avx2);
  ------------------
  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  191|       |
  192|  8.57k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL))
  ------------------
  |  Branch (192:9): [True: 8.57k, False: 0]
  ------------------
  193|  8.57k|        return;
  194|       |
  195|  8.57k|    init_8tap_fns(avx512icl);
  ------------------
  |  |  143|      0|    init_8tap_gen(mc,  opt); \
  |  |  ------------------
  |  |  |  |  132|      0|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|      0|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|      0|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|      0|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|      0|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|      0|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|      0|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|      0|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   36|  8.57k|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  |  |  144|      0|    init_8tap_gen(mct, opt)
  |  |  ------------------
  |  |  |  |  132|      0|    init_##name##_fn(FILTER_2D_8TAP_REGULAR,        8tap_regular,        opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  133|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SMOOTH, 8tap_regular_smooth, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  134|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_REGULAR_SHARP,  8tap_regular_sharp,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  135|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_REGULAR, 8tap_smooth_regular, opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  136|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH,         8tap_smooth,         opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  137|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SMOOTH_SHARP,   8tap_smooth_sharp,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  138|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_REGULAR,  8tap_sharp_regular,  opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  139|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP_SMOOTH,   8tap_sharp_smooth,   opt); \
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  |  |  140|  8.57k|    init_##name##_fn(FILTER_2D_8TAP_SHARP,          8tap_sharp,          opt)
  |  |  |  |  ------------------
  |  |  |  |  |  |   38|  8.57k|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  |  |  |  |  ------------------
  |  |  |  |  |  |  |  |   52|  8.57k|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  |  |  |  |  ------------------
  |  |  |  |  ------------------
  |  |  ------------------
  ------------------
  196|       |
  197|      0|    init_mc_fn (FILTER_2D_BILINEAR,            bilin,               avx512icl);
  ------------------
  |  |   36|      0|    c->mc[type] = BF(dav1d_put_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  198|      0|    init_mct_fn(FILTER_2D_BILINEAR,            bilin,               avx512icl);
  ------------------
  |  |   38|      0|    c->mct[type] = BF(dav1d_prep_##name, suffix)
  |  |  ------------------
  |  |  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  |  |  ------------------
  ------------------
  199|       |
  200|      0|    c->avg = BF(dav1d_avg, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  201|      0|    c->w_avg = BF(dav1d_w_avg, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  202|      0|    c->mask = BF(dav1d_mask, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  203|      0|    c->w_mask[0] = BF(dav1d_w_mask_444, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  204|      0|    c->w_mask[1] = BF(dav1d_w_mask_422, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  205|      0|    c->w_mask[2] = BF(dav1d_w_mask_420, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  206|      0|    c->blend = BF(dav1d_blend, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  207|      0|    c->blend_v = BF(dav1d_blend_v, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  208|      0|    c->blend_h = BF(dav1d_blend_h, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  209|       |
  210|      0|    if (!(flags & DAV1D_X86_CPU_FLAG_SLOW_GATHER)) {
  ------------------
  |  Branch (210:9): [True: 0, False: 0]
  ------------------
  211|      0|        c->resize = BF(dav1d_resize, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  212|      0|        c->warp8x8  = BF(dav1d_warp_affine_8x8, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  213|      0|        c->warp8x8t = BF(dav1d_warp_affine_8x8t, avx512icl);
  ------------------
  |  |   52|      0|#define BF(x, suffix) x##_8bpc_##suffix
  ------------------
  214|      0|    }
  215|      0|#endif
  216|      0|}

msac.c:msac_init_x86:
   59|   315k|static ALWAYS_INLINE void msac_init_x86(MsacContext *const s) {
   60|   315k|    const unsigned flags = dav1d_get_cpu_flags();
   61|       |
   62|   315k|    if (flags & DAV1D_X86_CPU_FLAG_SSE2) {
  ------------------
  |  Branch (62:9): [True: 315k, False: 18.4E]
  ------------------
   63|   315k|        s->symbol_adapt16 = dav1d_msac_decode_symbol_adapt16_sse2;
   64|   315k|    }
   65|       |
   66|   315k|    if (flags & DAV1D_X86_CPU_FLAG_AVX2) {
  ------------------
  |  Branch (66:9): [True: 315k, False: 18.4E]
  ------------------
   67|   315k|        s->symbol_adapt16 = dav1d_msac_decode_symbol_adapt16_avx2;
   68|   315k|    }
   69|   315k|}

pal.c:pal_dsp_init_x86:
   34|  9.41k|static ALWAYS_INLINE void pal_dsp_init_x86(Dav1dPalDSPContext *const c) {
   35|  9.41k|    const unsigned flags = dav1d_get_cpu_flags();
   36|       |
   37|  9.41k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (37:9): [True: 0, False: 9.41k]
  ------------------
   38|       |
   39|  9.41k|    c->pal_idx_finish = dav1d_pal_idx_finish_ssse3;
   40|       |
   41|  9.41k|#if ARCH_X86_64
   42|  9.41k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (42:9): [True: 0, False: 9.41k]
  ------------------
   43|       |
   44|  9.41k|    c->pal_idx_finish = dav1d_pal_idx_finish_avx2;
   45|       |
   46|  9.41k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (46:9): [True: 9.41k, False: 0]
  ------------------
   47|       |
   48|      0|    c->pal_idx_finish = dav1d_pal_idx_finish_avx512icl;
   49|      0|#endif
   50|      0|}

refmvs.c:refmvs_dsp_init_x86:
   41|  9.41k|static ALWAYS_INLINE void refmvs_dsp_init_x86(Dav1dRefmvsDSPContext *const c) {
   42|  9.41k|    const unsigned flags = dav1d_get_cpu_flags();
   43|       |
   44|  9.41k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE2)) return;
  ------------------
  |  Branch (44:9): [True: 0, False: 9.41k]
  ------------------
   45|       |
   46|  9.41k|    c->splat_mv = dav1d_splat_mv_sse2;
   47|       |
   48|  9.41k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSSE3)) return;
  ------------------
  |  Branch (48:9): [True: 0, False: 9.41k]
  ------------------
   49|       |
   50|  9.41k|    c->save_tmvs = dav1d_save_tmvs_ssse3;
   51|       |
   52|  9.41k|    if (!(flags & DAV1D_X86_CPU_FLAG_SSE41)) return;
  ------------------
  |  Branch (52:9): [True: 0, False: 9.41k]
  ------------------
   53|  9.41k|#if ARCH_X86_64
   54|  9.41k|    c->load_tmvs = dav1d_load_tmvs_sse4;
   55|       |
   56|  9.41k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX2)) return;
  ------------------
  |  Branch (56:9): [True: 0, False: 9.41k]
  ------------------
   57|       |
   58|  9.41k|    c->save_tmvs = dav1d_save_tmvs_avx2;
   59|  9.41k|    c->splat_mv = dav1d_splat_mv_avx2;
   60|       |
   61|  9.41k|    if (!(flags & DAV1D_X86_CPU_FLAG_AVX512ICL)) return;
  ------------------
  |  Branch (61:9): [True: 9.41k, False: 0]
  ------------------
   62|       |
   63|      0|    c->save_tmvs = dav1d_save_tmvs_avx512icl;
   64|      0|    c->splat_mv = dav1d_splat_mv_avx512icl;
   65|      0|#endif
   66|      0|}

LLVMFuzzerInitialize:
   59|      2|int LLVMFuzzerInitialize(int *argc, char ***argv) {
   60|      2|    int i = 1;
   61|     11|    for (; i < *argc; i++) {
  ------------------
  |  Branch (61:12): [True: 9, False: 2]
  ------------------
   62|      9|        if (!strcmp((*argv)[i], "--cpumask")) {
  ------------------
  |  Branch (62:13): [True: 0, False: 9]
  ------------------
   63|      0|            const char * cpumask = (*argv)[i+1];
   64|      0|            if (cpumask) {
  ------------------
  |  Branch (64:17): [True: 0, False: 0]
  ------------------
   65|      0|                char *end;
   66|      0|                unsigned res;
   67|      0|                if (!strncmp(cpumask, "0x", 2)) {
  ------------------
  |  Branch (67:21): [True: 0, False: 0]
  ------------------
   68|      0|                    cpumask += 2;
   69|      0|                    res = (unsigned) strtoul(cpumask, &end, 16);
   70|      0|                } else {
   71|      0|                    res = (unsigned) strtoul(cpumask, &end, 0);
   72|      0|                }
   73|      0|                if (end != cpumask && !end[0]) {
  ------------------
  |  Branch (73:21): [True: 0, False: 0]
  |  Branch (73:39): [True: 0, False: 0]
  ------------------
   74|      0|                    dav1d_set_cpu_flags_mask(res);
   75|      0|                }
   76|      0|            }
   77|      0|            break;
   78|      0|        }
   79|      9|    }
   80|       |
   81|      2|    for (; i < *argc - 2; i++) {
  ------------------
  |  Branch (81:12): [True: 0, False: 2]
  ------------------
   82|      0|        (*argv)[i] = (*argv)[i + 2];
   83|      0|    }
   84|       |
   85|      2|    *argc = i;
   86|       |
   87|      2|    return 0;
   88|      2|}
LLVMFuzzerTestOneInput:
   94|  9.42k|{
   95|  9.42k|    Dav1dSettings settings = { 0 };
   96|  9.42k|    Dav1dContext * ctx = NULL;
   97|  9.42k|    Dav1dPicture pic;
   98|  9.42k|    const uint8_t *ptr = data;
   99|  9.42k|    int have_seq_hdr = 0;
  100|  9.42k|    int err;
  101|       |
  102|  9.42k|    dav1d_version();
  103|       |
  104|  9.42k|    if (size < 32) goto end;
  ------------------
  |  Branch (104:9): [True: 8, False: 9.41k]
  ------------------
  105|       |#ifdef DAV1D_ALLOC_FAIL
  106|       |    unsigned h = djb_xor(ptr, 32);
  107|       |    unsigned seed = h;
  108|       |    unsigned probability = h > (RAND_MAX >> 5) ? RAND_MAX >> 5 : h;
  109|       |    int max_frame_delay = (h & 0xf) + 1;
  110|       |    int n_threads = ((h >> 4) & 0x7) + 1;
  111|       |    if (max_frame_delay > 5) max_frame_delay = 1;
  112|       |    if (n_threads > 3) n_threads = 1;
  113|       |#endif
  114|  9.41k|    ptr += 32; // skip ivf header
  115|       |
  116|  9.41k|    dav1d_default_settings(&settings);
  117|       |
  118|  9.41k|#ifdef DAV1D_MT_FUZZING
  119|  9.41k|    settings.max_frame_delay = settings.n_threads = 4;
  120|       |#elif defined(DAV1D_ALLOC_FAIL)
  121|       |    settings.max_frame_delay = max_frame_delay;
  122|       |    settings.n_threads = n_threads;
  123|       |    dav1d_setup_alloc_fail(seed, probability);
  124|       |#else
  125|       |    settings.max_frame_delay = settings.n_threads = 1;
  126|       |#endif
  127|  9.41k|#if defined(DAV1D_FUZZ_MAX_SIZE)
  128|  9.41k|    settings.frame_size_limit = DAV1D_FUZZ_MAX_SIZE;
  ------------------
  |  |   56|  9.41k|#define DAV1D_FUZZ_MAX_SIZE 4096 * 4096
  ------------------
  129|  9.41k|#endif
  130|       |
  131|  9.41k|    err = dav1d_open(&ctx, &settings);
  132|  9.41k|    if (err < 0) goto end;
  ------------------
  |  Branch (132:9): [True: 0, False: 9.41k]
  ------------------
  133|       |
  134|   429k|    while (ptr <= data + size - 12) {
  ------------------
  |  Branch (134:12): [True: 423k, False: 6.09k]
  ------------------
  135|   423k|        Dav1dData buf;
  136|   423k|        uint8_t *p;
  137|       |
  138|   423k|        size_t frame_size = r32le(ptr);
  139|   423k|        ptr += 12;
  140|       |
  141|   423k|        if (frame_size > size || ptr > data + size - frame_size)
  ------------------
  |  Branch (141:13): [True: 2.69k, False: 420k]
  |  Branch (141:34): [True: 626, False: 419k]
  ------------------
  142|  3.32k|            break;
  143|       |
  144|   419k|        if (!frame_size) continue;
  ------------------
  |  Branch (144:13): [True: 1.83k, False: 417k]
  ------------------
  145|       |
  146|   417k|        if (!have_seq_hdr) {
  ------------------
  |  Branch (146:13): [True: 15.3k, False: 402k]
  ------------------
  147|  15.3k|            Dav1dSequenceHeader seq;
  148|  15.3k|            int err = dav1d_parse_sequence_header(&seq, ptr, frame_size);
  149|       |            // skip frames until we see a sequence header
  150|  15.3k|            if  (err != 0) {
  ------------------
  |  Branch (150:18): [True: 6.21k, False: 9.13k]
  ------------------
  151|  6.21k|                ptr += frame_size;
  152|  6.21k|                continue;
  153|  6.21k|            }
  154|  9.13k|            have_seq_hdr = 1;
  155|  9.13k|        }
  156|       |
  157|       |        // copy frame data to a new buffer to catch reads past the end of input
  158|   411k|        p = dav1d_data_create(&buf, frame_size);
  159|   411k|        if (!p) goto cleanup;
  ------------------
  |  Branch (159:13): [True: 0, False: 411k]
  ------------------
  160|   411k|        memcpy(p, ptr, frame_size);
  161|   411k|        ptr += frame_size;
  162|       |
  163|   430k|        do {
  164|   430k|            if ((err = dav1d_send_data(ctx, &buf)) < 0) {
  ------------------
  |  Branch (164:17): [True: 59.8k, False: 370k]
  ------------------
  165|  59.8k|                if (err != DAV1D_ERR(EAGAIN))
  ------------------
  |  |   58|  59.8k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (165:21): [True: 39.8k, False: 19.9k]
  ------------------
  166|  39.8k|                    break;
  167|  59.8k|            }
  168|   390k|            memset(&pic, 0, sizeof(pic));
  169|   390k|            err = dav1d_get_picture(ctx, &pic);
  170|   390k|            if (err == 0) {
  ------------------
  |  Branch (170:17): [True: 167k, False: 223k]
  ------------------
  171|   167k|                dav1d_picture_unref(&pic);
  172|   223k|            } else if (err != DAV1D_ERR(EAGAIN)) {
  ------------------
  |  |   58|   223k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (172:24): [True: 176k, False: 46.8k]
  ------------------
  173|   176k|                break;
  174|   176k|            }
  175|   390k|        } while (buf.sz > 0);
  ------------------
  |  Branch (175:18): [True: 18.5k, False: 195k]
  ------------------
  176|       |
  177|   411k|        if (buf.sz > 0)
  ------------------
  |  Branch (177:13): [True: 41.2k, False: 370k]
  ------------------
  178|  41.2k|            dav1d_data_unref(&buf);
  179|   411k|    }
  180|       |
  181|  9.41k|    memset(&pic, 0, sizeof(pic));
  182|  9.41k|    if ((err = dav1d_get_picture(ctx, &pic)) == 0) {
  ------------------
  |  Branch (182:9): [True: 5.16k, False: 4.25k]
  ------------------
  183|       |        /* Test calling dav1d_picture_unref() after dav1d_close() */
  184|  11.7k|        do {
  185|  11.7k|            Dav1dPicture pic2 = { 0 };
  186|  11.7k|            if ((err = dav1d_get_picture(ctx, &pic2)) == 0)
  ------------------
  |  Branch (186:17): [True: 4.25k, False: 7.45k]
  ------------------
  187|  4.25k|                dav1d_picture_unref(&pic2);
  188|  11.7k|        } while (err != DAV1D_ERR(EAGAIN));
  ------------------
  |  |   58|  11.7k|#define DAV1D_ERR(e) (-(e)) ///< Negate POSIX error code.
  ------------------
  |  Branch (188:18): [True: 6.54k, False: 5.16k]
  ------------------
  189|       |
  190|  5.16k|        dav1d_close(&ctx);
  191|  5.16k|        dav1d_picture_unref(&pic);
  192|  5.16k|        return 0;
  193|  5.16k|    }
  194|       |
  195|  4.25k|cleanup:
  196|  4.25k|    dav1d_close(&ctx);
  197|  4.26k|end:
  198|  4.26k|    return 0;
  199|  4.25k|}
dav1d_fuzzer.c:r32le:
   52|   423k|static unsigned r32le(const uint8_t *const p) {
   53|   423k|    return ((uint32_t)p[3] << 24U) | (p[2] << 16U) | (p[1] << 8U) | p[0];
   54|   423k|}

